2018-02-06 07:52:58 +00:00
|
|
|
// © 2018 and later: Unicode, Inc. and others.
|
|
|
|
// License & terms of use: http://www.unicode.org/copyright.html
|
|
|
|
|
|
|
|
#include "unicode/utypes.h"
|
|
|
|
|
2018-04-23 23:02:26 +00:00
|
|
|
#if !UCONFIG_NO_FORMATTING
|
2018-02-06 07:52:58 +00:00
|
|
|
|
2018-02-13 02:23:52 +00:00
|
|
|
// Allow implicit conversion from char16_t* to UnicodeString for this file:
|
|
|
|
// Helpful in toString methods and elsewhere.
|
|
|
|
#define UNISTR_FROM_STRING_EXPLICIT
|
|
|
|
|
2018-02-06 07:52:58 +00:00
|
|
|
#include "numparse_types.h"
|
|
|
|
#include "numparse_stringsegment.h"
|
|
|
|
#include "putilimp.h"
|
|
|
|
#include "unicode/utf16.h"
|
2018-02-08 09:59:35 +00:00
|
|
|
#include "unicode/uniset.h"
|
2018-02-06 07:52:58 +00:00
|
|
|
|
|
|
|
using namespace icu;
|
|
|
|
using namespace icu::numparse;
|
|
|
|
using namespace icu::numparse::impl;
|
|
|
|
|
|
|
|
|
2018-03-23 06:46:19 +00:00
|
|
|
StringSegment::StringSegment(const UnicodeString& str, bool ignoreCase)
|
2018-02-08 09:59:35 +00:00
|
|
|
: fStr(str), fStart(0), fEnd(str.length()),
|
2018-03-23 06:46:19 +00:00
|
|
|
fFoldCase(ignoreCase) {}
|
2018-02-06 07:52:58 +00:00
|
|
|
|
|
|
|
int32_t StringSegment::getOffset() const {
|
|
|
|
return fStart;
|
|
|
|
}
|
|
|
|
|
|
|
|
void StringSegment::setOffset(int32_t start) {
|
|
|
|
fStart = start;
|
|
|
|
}
|
|
|
|
|
|
|
|
void StringSegment::adjustOffset(int32_t delta) {
|
|
|
|
fStart += delta;
|
|
|
|
}
|
|
|
|
|
2018-02-08 09:59:35 +00:00
|
|
|
void StringSegment::adjustOffsetByCodePoint() {
|
|
|
|
fStart += U16_LENGTH(getCodePoint());
|
|
|
|
}
|
|
|
|
|
2018-02-06 07:52:58 +00:00
|
|
|
void StringSegment::setLength(int32_t length) {
|
|
|
|
fEnd = fStart + length;
|
|
|
|
}
|
|
|
|
|
|
|
|
void StringSegment::resetLength() {
|
|
|
|
fEnd = fStr.length();
|
|
|
|
}
|
|
|
|
|
|
|
|
int32_t StringSegment::length() const {
|
|
|
|
return fEnd - fStart;
|
|
|
|
}
|
|
|
|
|
|
|
|
char16_t StringSegment::charAt(int32_t index) const {
|
|
|
|
return fStr.charAt(index + fStart);
|
|
|
|
}
|
|
|
|
|
|
|
|
UChar32 StringSegment::codePointAt(int32_t index) const {
|
|
|
|
return fStr.char32At(index + fStart);
|
|
|
|
}
|
|
|
|
|
|
|
|
UnicodeString StringSegment::toUnicodeString() const {
|
2018-03-28 03:42:12 +00:00
|
|
|
return UnicodeString(fStr.getBuffer() + fStart, fEnd - fStart);
|
|
|
|
}
|
|
|
|
|
|
|
|
const UnicodeString StringSegment::toTempUnicodeString() const {
|
2018-03-23 10:07:38 +00:00
|
|
|
// Use the readonly-aliasing constructor for efficiency.
|
|
|
|
return UnicodeString(FALSE, fStr.getBuffer() + fStart, fEnd - fStart);
|
2018-02-06 07:52:58 +00:00
|
|
|
}
|
|
|
|
|
|
|
|
UChar32 StringSegment::getCodePoint() const {
|
|
|
|
char16_t lead = fStr.charAt(fStart);
|
|
|
|
if (U16_IS_LEAD(lead) && fStart + 1 < fEnd) {
|
|
|
|
return fStr.char32At(fStart);
|
|
|
|
} else if (U16_IS_SURROGATE(lead)) {
|
|
|
|
return -1;
|
|
|
|
} else {
|
|
|
|
return lead;
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2018-03-21 06:30:29 +00:00
|
|
|
bool StringSegment::startsWith(UChar32 otherCp) const {
|
2018-02-08 09:59:35 +00:00
|
|
|
return codePointsEqual(getCodePoint(), otherCp, fFoldCase);
|
|
|
|
}
|
|
|
|
|
2018-03-21 06:30:29 +00:00
|
|
|
bool StringSegment::startsWith(const UnicodeSet& uniset) const {
|
2018-02-08 09:59:35 +00:00
|
|
|
// TODO: Move UnicodeSet case-folding logic here.
|
|
|
|
// TODO: Handle string matches here instead of separately.
|
|
|
|
UChar32 cp = getCodePoint();
|
|
|
|
if (cp == -1) {
|
|
|
|
return false;
|
|
|
|
}
|
|
|
|
return uniset.contains(cp);
|
|
|
|
}
|
|
|
|
|
2018-03-21 06:30:29 +00:00
|
|
|
bool StringSegment::startsWith(const UnicodeString& other) const {
|
|
|
|
if (other.isBogus() || other.length() == 0 || length() == 0) {
|
|
|
|
return false;
|
|
|
|
}
|
|
|
|
int cp1 = getCodePoint();
|
|
|
|
int cp2 = other.char32At(0);
|
|
|
|
return codePointsEqual(cp1, cp2, fFoldCase);
|
|
|
|
}
|
|
|
|
|
2018-02-08 09:59:35 +00:00
|
|
|
int32_t StringSegment::getCommonPrefixLength(const UnicodeString& other) {
|
|
|
|
return getPrefixLengthInternal(other, fFoldCase);
|
|
|
|
}
|
|
|
|
|
|
|
|
int32_t StringSegment::getCaseSensitivePrefixLength(const UnicodeString& other) {
|
|
|
|
return getPrefixLengthInternal(other, false);
|
|
|
|
}
|
|
|
|
|
|
|
|
int32_t StringSegment::getPrefixLengthInternal(const UnicodeString& other, bool foldCase) {
|
2018-02-06 07:52:58 +00:00
|
|
|
int32_t offset = 0;
|
|
|
|
for (; offset < uprv_min(length(), other.length());) {
|
2018-02-08 09:59:35 +00:00
|
|
|
// TODO: case-fold code points, not chars
|
|
|
|
char16_t c1 = charAt(offset);
|
|
|
|
char16_t c2 = other.charAt(offset);
|
|
|
|
if (!codePointsEqual(c1, c2, foldCase)) {
|
2018-02-06 07:52:58 +00:00
|
|
|
break;
|
|
|
|
}
|
|
|
|
offset++;
|
|
|
|
}
|
|
|
|
return offset;
|
|
|
|
}
|
|
|
|
|
2018-02-08 09:59:35 +00:00
|
|
|
bool StringSegment::codePointsEqual(UChar32 cp1, UChar32 cp2, bool foldCase) {
|
|
|
|
if (cp1 == cp2) {
|
|
|
|
return true;
|
|
|
|
}
|
|
|
|
if (!foldCase) {
|
|
|
|
return false;
|
|
|
|
}
|
|
|
|
cp1 = u_foldCase(cp1, TRUE);
|
|
|
|
cp2 = u_foldCase(cp2, TRUE);
|
|
|
|
return cp1 == cp2;
|
|
|
|
}
|
|
|
|
|
2018-03-24 05:46:28 +00:00
|
|
|
bool StringSegment::operator==(const UnicodeString& other) const {
|
2018-03-28 03:42:12 +00:00
|
|
|
return toTempUnicodeString() == other;
|
2018-03-24 05:46:28 +00:00
|
|
|
}
|
|
|
|
|
2018-02-06 07:52:58 +00:00
|
|
|
|
|
|
|
#endif /* #if !UCONFIG_NO_FORMATTING */
|