scuffed-code/icu4c/source/i18n/strmatch.cpp

/*
* Copyright (C) 2001, International Business Machines Corporation and others. All Rights Reserved.
**********************************************************************
*   Date        Name        Description
*   07/23/01    aliu        Creation.
**********************************************************************
*/

#include "strmatch.h"
#include "rbt_data.h"
#include "rbt_rule.h"

U_NAMESPACE_BEGIN

StringMatcher::StringMatcher(const UnicodeString& theString,
                             int32_t start,
                             int32_t limit,
                             UBool isSeg,
                             const TransliterationRuleData& theData) :
    data(theData),
    isSegment(isSeg),
    matchStart(-1),
    matchLimit(-1)
{
    theString.extractBetween(start, limit, pattern);
}

StringMatcher::StringMatcher(const StringMatcher& o) :
    UnicodeMatcher(o),
    pattern(o.pattern),
    data(o.data),
    isSegment(o.isSegment),
    matchStart(o.matchStart),
    matchLimit(o.matchStart)
{
}

/**
 * Destructor
 */
StringMatcher::~StringMatcher() {
}

/**
 * Implement UnicodeMatcher
 */
UnicodeMatcher* StringMatcher::clone() const {
    return new StringMatcher(*this);
}

/**
 * Implement UnicodeMatcher
 */
UMatchDegree StringMatcher::matches(const Replaceable& text,
                                    int32_t& offset,
                                    int32_t limit,
                                    UBool incremental) {
    int32_t i;
    int32_t cursor = offset;
    if (limit < cursor) {
        // Match in the reverse direction
        for (i=pattern.length()-1; i>=0; --i) {
            UChar keyChar = pattern.charAt(i);
            UnicodeMatcher* subm = data.lookup(keyChar);
            if (subm == 0) {
                if (cursor >= limit &&
                    keyChar == text.charAt(cursor)) {
                    --cursor;
                } else {
                    return U_MISMATCH;
                }
            } else {
                UMatchDegree m =
                    subm->matches(text, cursor, limit, incremental);
                if (m != U_MATCH) {
                    return m;
                }
            }
        }
        // Record the match position, but adjust for a normal
        // forward start, limit, and only if a prior match does not
        // exist -- we want the rightmost match.
        if (matchStart < 0) {
            matchStart = cursor+1;
            matchLimit = offset+1;
        }
    } else {
        for (i=0; i<pattern.length(); ++i) {
            if (incremental && cursor == limit) {
                // We've reached the context limit without a mismatch and
                // without completing our match.
                return U_PARTIAL_MATCH;
            }
            UChar keyChar = pattern.charAt(i);
            UnicodeMatcher* subm = data.lookup(keyChar);
            if (subm == 0) {
                // Don't need the cursor < limit check if
                // incremental is TRUE (because it's done above); do need
                // it otherwise.
                if (cursor < limit &&
                    keyChar == text.charAt(cursor)) {
                    ++cursor;
                } else {
                    return U_MISMATCH;
                }
            } else {
                UMatchDegree m =
                    subm->matches(text, cursor, limit, incremental);
                if (m != U_MATCH) {
                    return m;
                }
            }
        }
        // Record the match position
        matchStart = offset;
        matchLimit = cursor;
    }

    offset = cursor;
    return U_MATCH;
}

/**
 * Implement UnicodeMatcher
 */
UnicodeString& StringMatcher::toPattern(UnicodeString& result,
                                        UBool escapeUnprintable) const {
    UnicodeString str, quoteBuf;
    if (isSegment) {
        result.append((UChar)40); /*(*/
    }
    for (int32_t i=0; i<pattern.length(); ++i) {
        UChar keyChar = pattern.charAt(i);
        const UnicodeMatcher* m = data.lookup(keyChar);
        if (m == 0) {
            TransliterationRule::appendToRule(result, keyChar, FALSE, escapeUnprintable, quoteBuf);
        } else {
            TransliterationRule::appendToRule(result, m->toPattern(str, escapeUnprintable),
                         TRUE, escapeUnprintable, quoteBuf);
        }
    }
    if (isSegment) {
        result.append((UChar)41); /*)*/
    }
    // Flush quoteBuf out to result
    TransliterationRule::appendToRule(result, -1,
                                      TRUE, escapeUnprintable, quoteBuf);
    return result;
}

/**
 * Implement UnicodeMatcher
 */
UBool StringMatcher::matchesIndexValue(uint8_t v) const {
    if (pattern.length() == 0) {
        return TRUE;
    }
    UChar32 c = pattern.char32At(0);
    const UnicodeMatcher *m = data.lookup(c);
    return (m == 0) ? ((c & 0xFF) == v) : m->matchesIndexValue(v);
}

/**
 * Remove any match data.  This must be called before performing a
 * set of matches with this segment.
 */
 void StringMatcher::resetMatch() {
    matchStart = matchLimit = -1;
}

/**
 * Return the start offset, in the match text, of the <em>rightmost</em>
 * match.  This method may get moved up into the UnicodeMatcher if
 * it turns out to be useful to generalize this.
 */
int32_t StringMatcher::getMatchStart() const {
    return matchStart;
}

/**
 * Return the limit offset, in the match text, of the <em>rightmost</em>
 * match.  This method may get moved up into the UnicodeMatcher if
 * it turns out to be useful to generalize this.
 */
int32_t StringMatcher::getMatchLimit() const {
    return matchLimit;
}

U_NAMESPACE_END

//eof
ICU-1076 initial limited support for Kleene star and plus operators X-SVN-Rev: 5359 2001-07-27 00:18:53 +00:00			`/*`
			`* Copyright (C) 2001, International Business Machines Corporation and others. All Rights Reserved.`
			`**********************************************************************`
			`* Date Name Description`
			`* 07/23/01 aliu Creation.`
			`**********************************************************************`
			`*/`

			`#include "strmatch.h"`
			`#include "rbt_data.h"`
ICU-1076 implement toPattern X-SVN-Rev: 5379 2001-07-30 23:23:16 +00:00			`#include "rbt_rule.h"`
ICU-1076 initial limited support for Kleene star and plus operators X-SVN-Rev: 5359 2001-07-27 00:18:53 +00:00
ICU-1264 added namspace support where possible. X-SVN-Rev: 6124 2001-10-08 23:26:58 +00:00			`U_NAMESPACE_BEGIN`

ICU-1076 initial limited support for Kleene star and plus operators X-SVN-Rev: 5359 2001-07-27 00:18:53 +00:00			`StringMatcher::StringMatcher(const UnicodeString& theString,`
			`int32_t start,`
			`int32_t limit,`
ICU-1076 implement toPattern X-SVN-Rev: 5379 2001-07-30 23:23:16 +00:00			`UBool isSeg,`
ICU-1076 initial limited support for Kleene star and plus operators X-SVN-Rev: 5359 2001-07-27 00:18:53 +00:00			`const TransliterationRuleData& theData) :`
ICU-1076 implement toPattern X-SVN-Rev: 5379 2001-07-30 23:23:16 +00:00			`data(theData),`
ICU-1406 make quantified segments behave like perl counterparts X-SVN-Rev: 6493 2001-10-30 18:08:53 +00:00			`isSegment(isSeg),`
			`matchStart(-1),`
			`matchLimit(-1)`
ICU-900 Fixed some compiler warnings. X-SVN-Rev: 6136 2001-10-09 22:21:01 +00:00			`{`
ICU-1076 initial limited support for Kleene star and plus operators X-SVN-Rev: 5359 2001-07-27 00:18:53 +00:00			`theString.extractBetween(start, limit, pattern);`
			`}`

			`StringMatcher::StringMatcher(const StringMatcher& o) :`
ICU-900 Fixed some compiler warnings. X-SVN-Rev: 6136 2001-10-09 22:21:01 +00:00			`UnicodeMatcher(o),`
ICU-1076 initial limited support for Kleene star and plus operators X-SVN-Rev: 5359 2001-07-27 00:18:53 +00:00			`pattern(o.pattern),`
ICU-900 Fixed some compiler warnings. X-SVN-Rev: 6136 2001-10-09 22:21:01 +00:00			`data(o.data),`
ICU-1406 make quantified segments behave like perl counterparts X-SVN-Rev: 6493 2001-10-30 18:08:53 +00:00			`isSegment(o.isSegment),`
			`matchStart(o.matchStart),`
			`matchLimit(o.matchStart)`
ICU-900 Fixed some compiler warnings. X-SVN-Rev: 6136 2001-10-09 22:21:01 +00:00			`{`
ICU-1076 initial limited support for Kleene star and plus operators X-SVN-Rev: 5359 2001-07-27 00:18:53 +00:00			`}`

			`/**`
			`* Destructor`
			`*/`
			`StringMatcher::~StringMatcher() {`
			`}`

			`/**`
			`* Implement UnicodeMatcher`
			`*/`
			`UnicodeMatcher* StringMatcher::clone() const {`
			`return new StringMatcher(*this);`
			`}`

			`/**`
			`* Implement UnicodeMatcher`
			`*/`
			`UMatchDegree StringMatcher::matches(const Replaceable& text,`
			`int32_t& offset,`
			`int32_t limit,`
ICU-1406 make UnicodeMatcher::matches non-const X-SVN-Rev: 6503 2001-10-30 23:55:09 +00:00			`UBool incremental) {`
ICU-1076 initial limited support for Kleene star and plus operators X-SVN-Rev: 5359 2001-07-27 00:18:53 +00:00			`int32_t i;`
			`int32_t cursor = offset;`
			`if (limit < cursor) {`
ICU-1406 make quantified segments behave like perl counterparts X-SVN-Rev: 6493 2001-10-30 18:08:53 +00:00			`// Match in the reverse direction`
ICU-1076 initial limited support for Kleene star and plus operators X-SVN-Rev: 5359 2001-07-27 00:18:53 +00:00			`for (i=pattern.length()-1; i>=0; --i) {`
			`UChar keyChar = pattern.charAt(i);`
ICU-1406 make UnicodeMatcher::matches non-const X-SVN-Rev: 6503 2001-10-30 23:55:09 +00:00			`UnicodeMatcher* subm = data.lookup(keyChar);`
ICU-1076 initial limited support for Kleene star and plus operators X-SVN-Rev: 5359 2001-07-27 00:18:53 +00:00			`if (subm == 0) {`
			`if (cursor >= limit &&`
			`keyChar == text.charAt(cursor)) {`
			`--cursor;`
			`} else {`
			`return U_MISMATCH;`
			`}`
			`} else {`
			`UMatchDegree m =`
			`subm->matches(text, cursor, limit, incremental);`
			`if (m != U_MATCH) {`
			`return m;`
			`}`
			`}`
			`}`
ICU-1406 make quantified segments behave like perl counterparts X-SVN-Rev: 6493 2001-10-30 18:08:53 +00:00			`// Record the match position, but adjust for a normal`
			`// forward start, limit, and only if a prior match does not`
			`// exist -- we want the rightmost match.`
			`if (matchStart < 0) {`
ICU-1406 make UnicodeMatcher::matches non-const X-SVN-Rev: 6503 2001-10-30 23:55:09 +00:00			`matchStart = cursor+1;`
			`matchLimit = offset+1;`
ICU-1406 make quantified segments behave like perl counterparts X-SVN-Rev: 6493 2001-10-30 18:08:53 +00:00			`}`
ICU-1076 initial limited support for Kleene star and plus operators X-SVN-Rev: 5359 2001-07-27 00:18:53 +00:00			`} else {`
			`for (i=0; i<pattern.length(); ++i) {`
			`if (incremental && cursor == limit) {`
			`// We've reached the context limit without a mismatch and`
			`// without completing our match.`
			`return U_PARTIAL_MATCH;`
			`}`
			`UChar keyChar = pattern.charAt(i);`
ICU-1406 make UnicodeMatcher::matches non-const X-SVN-Rev: 6503 2001-10-30 23:55:09 +00:00			`UnicodeMatcher* subm = data.lookup(keyChar);`
ICU-1076 initial limited support for Kleene star and plus operators X-SVN-Rev: 5359 2001-07-27 00:18:53 +00:00			`if (subm == 0) {`
			`// Don't need the cursor < limit check if`
			`// incremental is TRUE (because it's done above); do need`
			`// it otherwise.`
			`if (cursor < limit &&`
			`keyChar == text.charAt(cursor)) {`
			`++cursor;`
			`} else {`
			`return U_MISMATCH;`
			`}`
			`} else {`
			`UMatchDegree m =`
			`subm->matches(text, cursor, limit, incremental);`
			`if (m != U_MATCH) {`
			`return m;`
			`}`
			`}`
			`}`
ICU-1406 make quantified segments behave like perl counterparts X-SVN-Rev: 6493 2001-10-30 18:08:53 +00:00			`// Record the match position`
ICU-1406 make UnicodeMatcher::matches non-const X-SVN-Rev: 6503 2001-10-30 23:55:09 +00:00			`matchStart = offset;`
			`matchLimit = cursor;`
ICU-1076 initial limited support for Kleene star and plus operators X-SVN-Rev: 5359 2001-07-27 00:18:53 +00:00			`}`

			`offset = cursor;`
			`return U_MATCH;`
			`}`

			`/**`
			`* Implement UnicodeMatcher`
			`*/`
			`UnicodeString& StringMatcher::toPattern(UnicodeString& result,`
			`UBool escapeUnprintable) const {`
ICU-1076 implement toPattern X-SVN-Rev: 5379 2001-07-30 23:23:16 +00:00			`UnicodeString str, quoteBuf;`
			`if (isSegment) {`
			`result.append((UChar)40); /(/`
			`}`
ICU-1076 initial limited support for Kleene star and plus operators X-SVN-Rev: 5359 2001-07-27 00:18:53 +00:00			`for (int32_t i=0; i<pattern.length(); ++i) {`
ICU-1076 implement toPattern X-SVN-Rev: 5379 2001-07-30 23:23:16 +00:00			`UChar keyChar = pattern.charAt(i);`
			`const UnicodeMatcher* m = data.lookup(keyChar);`
			`if (m == 0) {`
			`TransliterationRule::appendToRule(result, keyChar, FALSE, escapeUnprintable, quoteBuf);`
			`} else {`
			`TransliterationRule::appendToRule(result, m->toPattern(str, escapeUnprintable),`
			`TRUE, escapeUnprintable, quoteBuf);`
			`}`
			`}`
			`if (isSegment) {`
			`result.append((UChar)41); /)/`
ICU-1076 initial limited support for Kleene star and plus operators X-SVN-Rev: 5359 2001-07-27 00:18:53 +00:00			`}`
ICU-1076 implement toPattern X-SVN-Rev: 5379 2001-07-30 23:23:16 +00:00			`// Flush quoteBuf out to result`
ICU-1406 make quantified segments behave like perl counterparts X-SVN-Rev: 6493 2001-10-30 18:08:53 +00:00			`TransliterationRule::appendToRule(result, -1,`
			`TRUE, escapeUnprintable, quoteBuf);`
ICU-1076 initial limited support for Kleene star and plus operators X-SVN-Rev: 5359 2001-07-27 00:18:53 +00:00			`return result;`
			`}`

			`/**`
			`* Implement UnicodeMatcher`
			`*/`
			`UBool StringMatcher::matchesIndexValue(uint8_t v) const {`
			`if (pattern.length() == 0) {`
			`return TRUE;`
			`}`
			`UChar32 c = pattern.char32At(0);`
			`const UnicodeMatcher *m = data.lookup(c);`
			`return (m == 0) ? ((c & 0xFF) == v) : m->matchesIndexValue(v);`
			`}`

ICU-1406 make quantified segments behave like perl counterparts X-SVN-Rev: 6493 2001-10-30 18:08:53 +00:00			`/**`
			`* Remove any match data. This must be called before performing a`
			`* set of matches with this segment.`
			`*/`
			`void StringMatcher::resetMatch() {`
			`matchStart = matchLimit = -1;`
			`}`

			`/**`
			`* Return the start offset, in the match text, of the <em>rightmost</em>`
			`* match. This method may get moved up into the UnicodeMatcher if`
			`* it turns out to be useful to generalize this.`
			`*/`
			`int32_t StringMatcher::getMatchStart() const {`
			`return matchStart;`
			`}`

			`/**`
			`* Return the limit offset, in the match text, of the <em>rightmost</em>`
			`* match. This method may get moved up into the UnicodeMatcher if`
			`* it turns out to be useful to generalize this.`
			`*/`
			`int32_t StringMatcher::getMatchLimit() const {`
			`return matchLimit;`
			`}`

ICU-1264 added namspace support where possible. X-SVN-Rev: 6124 2001-10-08 23:26:58 +00:00			`U_NAMESPACE_END`

ICU-1076 initial limited support for Kleene star and plus operators X-SVN-Rev: 5359 2001-07-27 00:18:53 +00:00			`//eof`
ICU-1264 added namspace support where possible. X-SVN-Rev: 6124 2001-10-08 23:26:58 +00:00