scuffed-code/icu4c/source/i18n/numparse_stringsegment.cpp

// © 2018 and later: Unicode, Inc. and others.
// License & terms of use: http://www.unicode.org/copyright.html

#include "unicode/utypes.h"

#if !UCONFIG_NO_FORMATTING

// Allow implicit conversion from char16_t* to UnicodeString for this file:
// Helpful in toString methods and elsewhere.
#define UNISTR_FROM_STRING_EXPLICIT

#include "numparse_types.h"
#include "numparse_stringsegment.h"
#include "putilimp.h"
#include "unicode/utf16.h"
#include "unicode/uniset.h"

using namespace icu;
using namespace icu::numparse;
using namespace icu::numparse::impl;


StringSegment::StringSegment(const UnicodeString& str, bool ignoreCase)
        : fStr(str), fStart(0), fEnd(str.length()),
          fFoldCase(ignoreCase) {}

int32_t StringSegment::getOffset() const {
    return fStart;
}

void StringSegment::setOffset(int32_t start) {
    fStart = start;
}

void StringSegment::adjustOffset(int32_t delta) {
    fStart += delta;
}

void StringSegment::adjustOffsetByCodePoint() {
    fStart += U16_LENGTH(getCodePoint());
}

void StringSegment::setLength(int32_t length) {
    fEnd = fStart + length;
}

void StringSegment::resetLength() {
    fEnd = fStr.length();
}

int32_t StringSegment::length() const {
    return fEnd - fStart;
}

char16_t StringSegment::charAt(int32_t index) const {
    return fStr.charAt(index + fStart);
}

UChar32 StringSegment::codePointAt(int32_t index) const {
    return fStr.char32At(index + fStart);
}

UnicodeString StringSegment::toUnicodeString() const {
    return UnicodeString(fStr.getBuffer() + fStart, fEnd - fStart);
}

const UnicodeString StringSegment::toTempUnicodeString() const {
    // Use the readonly-aliasing constructor for efficiency.
    return UnicodeString(FALSE, fStr.getBuffer() + fStart, fEnd - fStart);
}

UChar32 StringSegment::getCodePoint() const {
    char16_t lead = fStr.charAt(fStart);
    if (U16_IS_LEAD(lead) && fStart + 1 < fEnd) {
        return fStr.char32At(fStart);
    } else if (U16_IS_SURROGATE(lead)) {
        return -1;
    } else {
        return lead;
    }
}

bool StringSegment::startsWith(UChar32 otherCp) const {
    return codePointsEqual(getCodePoint(), otherCp, fFoldCase);
}

bool StringSegment::startsWith(const UnicodeSet& uniset) const {
    // TODO: Move UnicodeSet case-folding logic here.
    // TODO: Handle string matches here instead of separately.
    UChar32 cp = getCodePoint();
    if (cp == -1) {
        return false;
    }
    return uniset.contains(cp);
}

bool StringSegment::startsWith(const UnicodeString& other) const {
    if (other.isBogus() || other.length() == 0 || length() == 0) {
        return false;
    }
    int cp1 = getCodePoint();
    int cp2 = other.char32At(0);
    return codePointsEqual(cp1, cp2, fFoldCase);
}

int32_t StringSegment::getCommonPrefixLength(const UnicodeString& other) {
    return getPrefixLengthInternal(other, fFoldCase);
}

int32_t StringSegment::getCaseSensitivePrefixLength(const UnicodeString& other) {
    return getPrefixLengthInternal(other, false);
}

int32_t StringSegment::getPrefixLengthInternal(const UnicodeString& other, bool foldCase) {
    U_ASSERT(other.length() > 0);
    int32_t offset = 0;
    for (; offset < uprv_min(length(), other.length());) {
        // TODO: case-fold code points, not chars
        char16_t c1 = charAt(offset);
        char16_t c2 = other.charAt(offset);
        if (!codePointsEqual(c1, c2, foldCase)) {
            break;
        }
        offset++;
    }
    return offset;
}

bool StringSegment::codePointsEqual(UChar32 cp1, UChar32 cp2, bool foldCase) {
    if (cp1 == cp2) {
        return true;
    }
    if (!foldCase) {
        return false;
    }
    cp1 = u_foldCase(cp1, TRUE);
    cp2 = u_foldCase(cp2, TRUE);
    return cp1 == cp2;
}

bool StringSegment::operator==(const UnicodeString& other) const {
    return toTempUnicodeString() == other;
}


#endif /* #if !UCONFIG_NO_FORMATTING */
ICU-13574 Porting the parsing utility classes StringSegment and UnicodeSetStaticCache to C++. X-SVN-Rev: 40841 2018-02-06 07:52:58 +00:00			`// © 2018 and later: Unicode, Inc. and others.`
			`// License & terms of use: http://www.unicode.org/copyright.html`

			`#include "unicode/utypes.h"`

ICU-13393 Removing the UPRV_INCOMPLETE_CPP11_SUPPORT flag since the number formatting code is no longer isolated from the rest of ICU. X-SVN-Rev: 41266 2018-04-23 23:02:26 +00:00			`#if !UCONFIG_NO_FORMATTING`
ICU-13574 Porting the parsing utility classes StringSegment and UnicodeSetStaticCache to C++. X-SVN-Rev: 40841 2018-02-06 07:52:58 +00:00
ICU-13574 AffixMatcher is working. All simple parsing tests are passing. X-SVN-Rev: 40903 2018-02-13 02:23:52 +00:00			`// Allow implicit conversion from char16_t* to UnicodeString for this file:`
			`// Helpful in toString methods and elsewhere.`
			`#define UNISTR_FROM_STRING_EXPLICIT`

ICU-13574 Porting the parsing utility classes StringSegment and UnicodeSetStaticCache to C++. X-SVN-Rev: 40841 2018-02-06 07:52:58 +00:00			`#include "numparse_types.h"`
			`#include "numparse_stringsegment.h"`
			`#include "putilimp.h"`
			`#include "unicode/utf16.h"`
ICU-13574 Basic parsing tests are passing on the pieces of code written so far, DecimalMatcher and MinusSignMatcher. X-SVN-Rev: 40872 2018-02-08 09:59:35 +00:00			`#include "unicode/uniset.h"`
ICU-13574 Porting the parsing utility classes StringSegment and UnicodeSetStaticCache to C++. X-SVN-Rev: 40841 2018-02-06 07:52:58 +00:00
			`using namespace icu;`
			`using namespace icu::numparse;`
			`using namespace icu::numparse::impl;`


ICU-8610 Dirty commit of C++ work so far. Probably does not build. X-SVN-Rev: 41142 2018-03-23 06:46:19 +00:00			`StringSegment::StringSegment(const UnicodeString& str, bool ignoreCase)`
ICU-13574 Basic parsing tests are passing on the pieces of code written so far, DecimalMatcher and MinusSignMatcher. X-SVN-Rev: 40872 2018-02-08 09:59:35 +00:00			`: fStr(str), fStart(0), fEnd(str.length()),`
ICU-8610 Dirty commit of C++ work so far. Probably does not build. X-SVN-Rev: 41142 2018-03-23 06:46:19 +00:00			`fFoldCase(ignoreCase) {}`
ICU-13574 Porting the parsing utility classes StringSegment and UnicodeSetStaticCache to C++. X-SVN-Rev: 40841 2018-02-06 07:52:58 +00:00
			`int32_t StringSegment::getOffset() const {`
			`return fStart;`
			`}`

			`void StringSegment::setOffset(int32_t start) {`
			`fStart = start;`
			`}`

			`void StringSegment::adjustOffset(int32_t delta) {`
			`fStart += delta;`
			`}`

ICU-13574 Basic parsing tests are passing on the pieces of code written so far, DecimalMatcher and MinusSignMatcher. X-SVN-Rev: 40872 2018-02-08 09:59:35 +00:00			`void StringSegment::adjustOffsetByCodePoint() {`
			`fStart += U16_LENGTH(getCodePoint());`
			`}`

ICU-13574 Porting the parsing utility classes StringSegment and UnicodeSetStaticCache to C++. X-SVN-Rev: 40841 2018-02-06 07:52:58 +00:00			`void StringSegment::setLength(int32_t length) {`
			`fEnd = fStart + length;`
			`}`

			`void StringSegment::resetLength() {`
			`fEnd = fStr.length();`
			`}`

			`int32_t StringSegment::length() const {`
			`return fEnd - fStart;`
			`}`

			`char16_t StringSegment::charAt(int32_t index) const {`
			`return fStr.charAt(index + fStart);`
			`}`

			`UChar32 StringSegment::codePointAt(int32_t index) const {`
			`return fStr.char32At(index + fStart);`
			`}`

			`UnicodeString StringSegment::toUnicodeString() const {`
ICU-13597 Fixing safety of toUnicodeString() readonly aliases by moving that behavior to a new method, toTempUnicodeString(). X-SVN-Rev: 41164 2018-03-28 03:42:12 +00:00			`return UnicodeString(fStr.getBuffer() + fStart, fEnd - fStart);`
			`}`

			`const UnicodeString StringSegment::toTempUnicodeString() const {`
ICU-8610 C++ number skeleton code is building. Testing is next. X-SVN-Rev: 41144 2018-03-23 10:07:38 +00:00			`// Use the readonly-aliasing constructor for efficiency.`
			`return UnicodeString(FALSE, fStr.getBuffer() + fStart, fEnd - fStart);`
ICU-13574 Porting the parsing utility classes StringSegment and UnicodeSetStaticCache to C++. X-SVN-Rev: 40841 2018-02-06 07:52:58 +00:00			`}`

			`UChar32 StringSegment::getCodePoint() const {`
			`char16_t lead = fStr.charAt(fStart);`
			`if (U16_IS_LEAD(lead) && fStart + 1 < fEnd) {`
			`return fStr.char32At(fStart);`
			`} else if (U16_IS_SURROGATE(lead)) {`
			`return -1;`
			`} else {`
			`return lead;`
			`}`
			`}`

ICU-13634 Changes NumberParseMatcher getLeadCodePoints() to smokeTest() in C++ and Java. The new method is more versatile and eliminates the requirement to maintain two code paths for "lead chars" and "no lead chars". X-SVN-Rev: 41131 2018-03-21 06:30:29 +00:00			`bool StringSegment::startsWith(UChar32 otherCp) const {`
ICU-13574 Basic parsing tests are passing on the pieces of code written so far, DecimalMatcher and MinusSignMatcher. X-SVN-Rev: 40872 2018-02-08 09:59:35 +00:00			`return codePointsEqual(getCodePoint(), otherCp, fFoldCase);`
			`}`

ICU-13634 Changes NumberParseMatcher getLeadCodePoints() to smokeTest() in C++ and Java. The new method is more versatile and eliminates the requirement to maintain two code paths for "lead chars" and "no lead chars". X-SVN-Rev: 41131 2018-03-21 06:30:29 +00:00			`bool StringSegment::startsWith(const UnicodeSet& uniset) const {`
ICU-13574 Basic parsing tests are passing on the pieces of code written so far, DecimalMatcher and MinusSignMatcher. X-SVN-Rev: 40872 2018-02-08 09:59:35 +00:00			`// TODO: Move UnicodeSet case-folding logic here.`
			`// TODO: Handle string matches here instead of separately.`
			`UChar32 cp = getCodePoint();`
			`if (cp == -1) {`
			`return false;`
			`}`
			`return uniset.contains(cp);`
			`}`

ICU-13634 Changes NumberParseMatcher getLeadCodePoints() to smokeTest() in C++ and Java. The new method is more versatile and eliminates the requirement to maintain two code paths for "lead chars" and "no lead chars". X-SVN-Rev: 41131 2018-03-21 06:30:29 +00:00			`bool StringSegment::startsWith(const UnicodeString& other) const {`
			`if (other.isBogus() \|\| other.length() == 0 \|\| length() == 0) {`
			`return false;`
			`}`
			`int cp1 = getCodePoint();`
			`int cp2 = other.char32At(0);`
			`return codePointsEqual(cp1, cp2, fFoldCase);`
			`}`

ICU-13574 Basic parsing tests are passing on the pieces of code written so far, DecimalMatcher and MinusSignMatcher. X-SVN-Rev: 40872 2018-02-08 09:59:35 +00:00			`int32_t StringSegment::getCommonPrefixLength(const UnicodeString& other) {`
			`return getPrefixLengthInternal(other, fFoldCase);`
			`}`

			`int32_t StringSegment::getCaseSensitivePrefixLength(const UnicodeString& other) {`
			`return getPrefixLengthInternal(other, false);`
			`}`

			`int32_t StringSegment::getPrefixLengthInternal(const UnicodeString& other, bool foldCase) {`
ICU-13804 Making number parsing code more robust when given empty symbol strings. X-SVN-Rev: 41497 2018-06-01 00:31:54 +00:00			`U_ASSERT(other.length() > 0);`
ICU-13574 Porting the parsing utility classes StringSegment and UnicodeSetStaticCache to C++. X-SVN-Rev: 40841 2018-02-06 07:52:58 +00:00			`int32_t offset = 0;`
			`for (; offset < uprv_min(length(), other.length());) {`
ICU-13574 Basic parsing tests are passing on the pieces of code written so far, DecimalMatcher and MinusSignMatcher. X-SVN-Rev: 40872 2018-02-08 09:59:35 +00:00			`// TODO: case-fold code points, not chars`
			`char16_t c1 = charAt(offset);`
			`char16_t c2 = other.charAt(offset);`
			`if (!codePointsEqual(c1, c2, foldCase)) {`
ICU-13574 Porting the parsing utility classes StringSegment and UnicodeSetStaticCache to C++. X-SVN-Rev: 40841 2018-02-06 07:52:58 +00:00			`break;`
			`}`
			`offset++;`
			`}`
			`return offset;`
			`}`

ICU-13574 Basic parsing tests are passing on the pieces of code written so far, DecimalMatcher and MinusSignMatcher. X-SVN-Rev: 40872 2018-02-08 09:59:35 +00:00			`bool StringSegment::codePointsEqual(UChar32 cp1, UChar32 cp2, bool foldCase) {`
			`if (cp1 == cp2) {`
			`return true;`
			`}`
			`if (!foldCase) {`
			`return false;`
			`}`
			`cp1 = u_foldCase(cp1, TRUE);`
			`cp2 = u_foldCase(cp2, TRUE);`
			`return cp1 == cp2;`
			`}`

ICU-8610 Adding tests for number skeletons in C++. Adding error code handling to the setToDecNumber setter on DecimalQuantity. Refactoring char-to-uchar conversion in skeleton implementation code. X-SVN-Rev: 41152 2018-03-24 05:46:28 +00:00			`bool StringSegment::operator==(const UnicodeString& other) const {`
ICU-13597 Fixing safety of toUnicodeString() readonly aliases by moving that behavior to a new method, toTempUnicodeString(). X-SVN-Rev: 41164 2018-03-28 03:42:12 +00:00			`return toTempUnicodeString() == other;`
ICU-8610 Adding tests for number skeletons in C++. Adding error code handling to the setToDecNumber setter on DecimalQuantity. Refactoring char-to-uchar conversion in skeleton implementation code. X-SVN-Rev: 41152 2018-03-24 05:46:28 +00:00			`}`

ICU-13574 Porting the parsing utility classes StringSegment and UnicodeSetStaticCache to C++. X-SVN-Rev: 40841 2018-02-06 07:52:58 +00:00
			`#endif /* #if !UCONFIG_NO_FORMATTING */`