File: autocomplete_i18n.h

package info (click to toggle)
chromium 138.0.7204.183-1~deb12u1
  • links: PTS, VCS
  • area: main
  • in suites: bookworm-proposed-updates
  • size: 6,080,960 kB
  • sloc: cpp: 34,937,079; ansic: 7,176,967; javascript: 4,110,704; python: 1,419,954; asm: 946,768; xml: 739,971; pascal: 187,324; sh: 89,623; perl: 88,663; objc: 79,944; sql: 50,304; cs: 41,786; fortran: 24,137; makefile: 21,811; php: 13,980; tcl: 13,166; yacc: 8,925; ruby: 7,485; awk: 3,720; lisp: 3,096; lex: 1,327; ada: 727; jsp: 228; sed: 36
file content (40 lines) | stat: -rw-r--r-- 1,853 bytes parent folder | download | duplicates (7)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
// Copyright 2015 The Chromium Authors
// Use of this source code is governed by a BSD-style license that can be
// found in the LICENSE file.

#ifndef COMPONENTS_OMNIBOX_BROWSER_AUTOCOMPLETE_I18N_H_
#define COMPONENTS_OMNIBOX_BROWSER_AUTOCOMPLETE_I18N_H_

#include "third_party/icu/source/common/unicode/uchar.h"

// Functor for a simple 16-bit Unicode case-insensitive comparison. This is
// designed for the autocomplete system where we would rather get prefix lenths
// correct than handle all possible case sensitivity issues.
//
// Any time this is used the result will be incorrect in some cases that
// certain users will be able to discern. Ideally, this class would be deleted
// and we would do full Unicode case-sensitivity mappings using
// base::i18n::ToLower. However, ToLower can change the lengths of strings,
// making computations of offsets or prefix lengths difficult. Getting all
// edge cases correct will require careful implementation and testing. In the
// mean time, we use this simpler approach.
//
// This comparator will not handle combining accents properly since it compares
// 16-bit values in isolation. If the two strings use the same sequence of
// combining accents (this is the normal case) in both strings, it will work.
//
// Additionally, this comparator does not decode UTF sequences which is why it
// is called "UCS2". UTF-16 surrogates will be compared literally (i.e. "case-
// sensitively").
//
// There are also a few cases where the lower-case version of a character
// expands to more than one code point that will not be handled properly. Such
// characters will be compared case-sensitively.
struct SimpleCaseInsensitiveCompareUCS2 {
 public:
  bool operator()(char16_t x, char16_t y) const {
    return u_tolower(x) == u_tolower(y);
  }
};

#endif  // COMPONENTS_OMNIBOX_BROWSER_AUTOCOMPLETE_I18N_H_