2018-02-10 06:36:07 +00:00
|
|
|
// © 2018 and later: Unicode, Inc. and others.
|
|
|
|
// License & terms of use: http://www.unicode.org/copyright.html
|
|
|
|
|
|
|
|
#include "unicode/utypes.h"
|
|
|
|
|
2018-04-23 23:02:26 +00:00
|
|
|
#if !UCONFIG_NO_FORMATTING
|
2018-02-10 06:36:07 +00:00
|
|
|
#ifndef __NUMPARSE_AFFIXES_H__
|
|
|
|
#define __NUMPARSE_AFFIXES_H__
|
|
|
|
|
|
|
|
#include "numparse_types.h"
|
2018-02-10 10:01:46 +00:00
|
|
|
#include "numparse_symbols.h"
|
|
|
|
#include "numparse_currency.h"
|
|
|
|
#include "number_affixutils.h"
|
2018-03-21 05:17:28 +00:00
|
|
|
#include "number_currencysymbols.h"
|
2018-02-10 06:36:07 +00:00
|
|
|
|
2018-02-13 02:23:52 +00:00
|
|
|
#include <array>
|
|
|
|
|
2018-02-10 10:01:46 +00:00
|
|
|
U_NAMESPACE_BEGIN namespace numparse {
|
2018-02-10 06:36:07 +00:00
|
|
|
namespace impl {
|
|
|
|
|
2018-02-10 10:01:46 +00:00
|
|
|
// Forward-declaration of implementation classes for friending
|
|
|
|
class AffixPatternMatcherBuilder;
|
|
|
|
class AffixPatternMatcher;
|
2018-02-10 06:36:07 +00:00
|
|
|
|
2018-02-10 14:29:26 +00:00
|
|
|
using ::icu::number::impl::AffixPatternProvider;
|
|
|
|
using ::icu::number::impl::TokenConsumer;
|
2018-03-21 05:17:28 +00:00
|
|
|
using ::icu::number::impl::CurrencySymbols;
|
2018-02-10 14:29:26 +00:00
|
|
|
|
2018-02-10 10:57:30 +00:00
|
|
|
|
|
|
|
class CodePointMatcher : public NumberParseMatcher, public UMemory {
|
2018-02-10 10:01:46 +00:00
|
|
|
public:
|
2018-02-10 10:57:30 +00:00
|
|
|
CodePointMatcher() = default; // WARNING: Leaves the object in an unusable state
|
|
|
|
|
|
|
|
CodePointMatcher(UChar32 cp);
|
|
|
|
|
|
|
|
bool match(StringSegment& segment, ParsedNumber& result, UErrorCode& status) const override;
|
|
|
|
|
2018-03-21 06:30:29 +00:00
|
|
|
bool smokeTest(const StringSegment& segment) const override;
|
2018-02-10 10:57:30 +00:00
|
|
|
|
2018-02-13 02:23:52 +00:00
|
|
|
UnicodeString toString() const override;
|
|
|
|
|
2018-02-10 10:57:30 +00:00
|
|
|
private:
|
|
|
|
UChar32 fCp;
|
|
|
|
};
|
|
|
|
|
|
|
|
|
2018-02-13 02:23:52 +00:00
|
|
|
/**
|
|
|
|
* A warehouse to retain ownership of CodePointMatchers.
|
|
|
|
*/
|
|
|
|
class CodePointMatcherWarehouse : public UMemory {
|
|
|
|
private:
|
|
|
|
static constexpr int32_t CODE_POINT_STACK_CAPACITY = 5; // Number of entries directly on the stack
|
|
|
|
static constexpr int32_t CODE_POINT_BATCH_SIZE = 10; // Number of entries per heap allocation
|
|
|
|
|
|
|
|
public:
|
|
|
|
CodePointMatcherWarehouse();
|
|
|
|
|
|
|
|
// A custom destructor is needed to free the memory from MaybeStackArray.
|
|
|
|
// A custom move constructor and move assignment seem to be needed because of the custom destructor.
|
|
|
|
|
|
|
|
~CodePointMatcherWarehouse();
|
|
|
|
|
|
|
|
CodePointMatcherWarehouse(CodePointMatcherWarehouse&& src) U_NOEXCEPT;
|
|
|
|
|
|
|
|
CodePointMatcherWarehouse& operator=(CodePointMatcherWarehouse&& src) U_NOEXCEPT;
|
|
|
|
|
|
|
|
NumberParseMatcher& nextCodePointMatcher(UChar32 cp);
|
|
|
|
|
|
|
|
private:
|
|
|
|
std::array<CodePointMatcher, CODE_POINT_STACK_CAPACITY> codePoints; // By value
|
|
|
|
MaybeStackArray<CodePointMatcher*, 3> codePointsOverflow; // On heap in "batches"
|
|
|
|
int32_t codePointCount; // Total for both the ones by value and on heap
|
|
|
|
int32_t codePointNumBatches; // Number of batches in codePointsOverflow
|
|
|
|
};
|
|
|
|
|
|
|
|
|
|
|
|
struct AffixTokenMatcherSetupData {
|
2018-03-21 05:17:28 +00:00
|
|
|
const CurrencySymbols& currencySymbols;
|
2018-02-13 02:23:52 +00:00
|
|
|
const DecimalFormatSymbols& dfs;
|
|
|
|
IgnorablesMatcher& ignorables;
|
|
|
|
const Locale& locale;
|
|
|
|
};
|
|
|
|
|
|
|
|
|
2018-02-10 10:57:30 +00:00
|
|
|
/**
|
|
|
|
* Small helper class that generates matchers for individual tokens for AffixPatternMatcher.
|
|
|
|
*
|
|
|
|
* In Java, this is called AffixTokenMatcherFactory (a "factory"). However, in C++, it is called a
|
|
|
|
* "warehouse", because in addition to generating the matchers, it also retains ownership of them. The
|
|
|
|
* warehouse must stay in scope for the whole lifespan of the AffixPatternMatcher that uses matchers from
|
|
|
|
* the warehouse.
|
|
|
|
*
|
|
|
|
* @author sffc
|
|
|
|
*/
|
2018-02-13 02:23:52 +00:00
|
|
|
class AffixTokenMatcherWarehouse : public UMemory {
|
2018-02-10 10:57:30 +00:00
|
|
|
public:
|
2018-02-10 14:29:26 +00:00
|
|
|
AffixTokenMatcherWarehouse() = default; // WARNING: Leaves the object in an unusable state
|
|
|
|
|
2018-02-13 02:23:52 +00:00
|
|
|
AffixTokenMatcherWarehouse(const AffixTokenMatcherSetupData* setupData);
|
2018-02-10 10:57:30 +00:00
|
|
|
|
2018-02-10 11:32:18 +00:00
|
|
|
NumberParseMatcher& minusSign();
|
|
|
|
|
|
|
|
NumberParseMatcher& plusSign();
|
|
|
|
|
|
|
|
NumberParseMatcher& percent();
|
|
|
|
|
|
|
|
NumberParseMatcher& permille();
|
|
|
|
|
|
|
|
NumberParseMatcher& currency(UErrorCode& status);
|
|
|
|
|
2018-02-13 02:23:52 +00:00
|
|
|
IgnorablesMatcher& ignorables();
|
|
|
|
|
2018-02-10 11:32:18 +00:00
|
|
|
NumberParseMatcher& nextCodePointMatcher(UChar32 cp);
|
2018-02-10 06:36:07 +00:00
|
|
|
|
2018-02-10 10:01:46 +00:00
|
|
|
private:
|
2018-02-13 02:23:52 +00:00
|
|
|
// NOTE: The following field may be unsafe to access after construction is done!
|
|
|
|
const AffixTokenMatcherSetupData* fSetupData;
|
2018-02-10 10:01:46 +00:00
|
|
|
|
|
|
|
// NOTE: These are default-constructed and should not be used until initialized.
|
2018-02-10 11:32:18 +00:00
|
|
|
MinusSignMatcher fMinusSign;
|
|
|
|
PlusSignMatcher fPlusSign;
|
|
|
|
PercentMatcher fPercent;
|
|
|
|
PermilleMatcher fPermille;
|
2018-03-31 05:18:51 +00:00
|
|
|
CombinedCurrencyMatcher fCurrency;
|
2018-02-10 10:01:46 +00:00
|
|
|
|
2018-02-13 02:23:52 +00:00
|
|
|
// Use a child class for code point matchers, since it requires non-default operators.
|
|
|
|
CodePointMatcherWarehouse fCodePoints;
|
2018-02-10 10:57:30 +00:00
|
|
|
|
2018-02-10 10:01:46 +00:00
|
|
|
friend class AffixPatternMatcherBuilder;
|
|
|
|
friend class AffixPatternMatcher;
|
|
|
|
};
|
|
|
|
|
|
|
|
|
2018-02-10 15:49:02 +00:00
|
|
|
class AffixPatternMatcherBuilder : public TokenConsumer, public MutableMatcherCollection {
|
2018-02-10 10:01:46 +00:00
|
|
|
public:
|
2018-02-10 10:57:30 +00:00
|
|
|
AffixPatternMatcherBuilder(const UnicodeString& pattern, AffixTokenMatcherWarehouse& warehouse,
|
2018-02-10 10:01:46 +00:00
|
|
|
IgnorablesMatcher* ignorables);
|
|
|
|
|
|
|
|
void consumeToken(::icu::number::impl::AffixPatternType type, UChar32 cp, UErrorCode& status) override;
|
|
|
|
|
|
|
|
/** NOTE: You can build only once! */
|
|
|
|
AffixPatternMatcher build();
|
|
|
|
|
|
|
|
private:
|
|
|
|
ArraySeriesMatcher::MatcherArray fMatchers;
|
|
|
|
int32_t fMatchersLen;
|
|
|
|
int32_t fLastTypeOrCp;
|
|
|
|
|
|
|
|
const UnicodeString& fPattern;
|
2018-02-10 10:57:30 +00:00
|
|
|
AffixTokenMatcherWarehouse& fWarehouse;
|
2018-02-10 10:01:46 +00:00
|
|
|
IgnorablesMatcher* fIgnorables;
|
|
|
|
|
2018-02-10 15:49:02 +00:00
|
|
|
void addMatcher(NumberParseMatcher& matcher) override;
|
2018-02-10 10:01:46 +00:00
|
|
|
};
|
|
|
|
|
|
|
|
|
|
|
|
class AffixPatternMatcher : public ArraySeriesMatcher {
|
|
|
|
public:
|
2018-02-10 14:29:26 +00:00
|
|
|
AffixPatternMatcher() = default; // WARNING: Leaves the object in an unusable state
|
|
|
|
|
2018-02-10 10:01:46 +00:00
|
|
|
static AffixPatternMatcher fromAffixPattern(const UnicodeString& affixPattern,
|
2018-02-10 10:57:30 +00:00
|
|
|
AffixTokenMatcherWarehouse& warehouse,
|
2018-02-10 10:01:46 +00:00
|
|
|
parse_flags_t parseFlags, bool* success,
|
|
|
|
UErrorCode& status);
|
|
|
|
|
2018-02-10 14:29:26 +00:00
|
|
|
UnicodeString getPattern() const;
|
2018-02-10 10:01:46 +00:00
|
|
|
|
2018-02-10 14:29:26 +00:00
|
|
|
bool operator==(const AffixPatternMatcher& other) const;
|
|
|
|
|
|
|
|
private:
|
|
|
|
CompactUnicodeString<4> fPattern;
|
2018-02-10 10:01:46 +00:00
|
|
|
|
2018-02-10 10:57:30 +00:00
|
|
|
AffixPatternMatcher(MatcherArray& matchers, int32_t matchersLen, const UnicodeString& pattern);
|
2018-02-10 10:01:46 +00:00
|
|
|
|
|
|
|
friend class AffixPatternMatcherBuilder;
|
|
|
|
};
|
2018-02-10 06:36:07 +00:00
|
|
|
|
|
|
|
|
2018-02-10 14:29:26 +00:00
|
|
|
class AffixMatcher : public NumberParseMatcher, public UMemory {
|
|
|
|
public:
|
|
|
|
AffixMatcher() = default; // WARNING: Leaves the object in an unusable state
|
|
|
|
|
|
|
|
AffixMatcher(AffixPatternMatcher* prefix, AffixPatternMatcher* suffix, result_flags_t flags);
|
|
|
|
|
|
|
|
bool match(StringSegment& segment, ParsedNumber& result, UErrorCode& status) const override;
|
|
|
|
|
|
|
|
void postProcess(ParsedNumber& result) const override;
|
|
|
|
|
2018-03-21 06:30:29 +00:00
|
|
|
bool smokeTest(const StringSegment& segment) const override;
|
2018-02-10 14:29:26 +00:00
|
|
|
|
2018-02-10 15:49:02 +00:00
|
|
|
int8_t compareTo(const AffixMatcher& rhs) const;
|
|
|
|
|
2018-02-13 02:23:52 +00:00
|
|
|
UnicodeString toString() const override;
|
|
|
|
|
2018-02-10 14:29:26 +00:00
|
|
|
private:
|
|
|
|
AffixPatternMatcher* fPrefix;
|
|
|
|
AffixPatternMatcher* fSuffix;
|
|
|
|
result_flags_t fFlags;
|
|
|
|
};
|
|
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
* A C++-only class to retain ownership of the AffixMatchers needed for parsing.
|
|
|
|
*/
|
|
|
|
class AffixMatcherWarehouse {
|
|
|
|
public:
|
|
|
|
AffixMatcherWarehouse() = default; // WARNING: Leaves the object in an unusable state
|
|
|
|
|
2018-02-13 02:23:52 +00:00
|
|
|
AffixMatcherWarehouse(AffixTokenMatcherWarehouse* tokenWarehouse);
|
2018-02-10 15:49:02 +00:00
|
|
|
|
2018-02-13 02:23:52 +00:00
|
|
|
void createAffixMatchers(const AffixPatternProvider& patternInfo, MutableMatcherCollection& output,
|
|
|
|
const IgnorablesMatcher& ignorables, parse_flags_t parseFlags,
|
|
|
|
UErrorCode& status);
|
2018-02-10 14:29:26 +00:00
|
|
|
|
|
|
|
private:
|
|
|
|
// 9 is the limit: positive, zero, and negative, each with prefix, suffix, and prefix+suffix
|
|
|
|
AffixMatcher fAffixMatchers[9];
|
|
|
|
// 6 is the limit: positive, zero, and negative, a prefix and a suffix for each
|
|
|
|
AffixPatternMatcher fAffixPatternMatchers[6];
|
2018-02-13 02:23:52 +00:00
|
|
|
// Reference to the warehouse for tokens used by the AffixPatternMatchers
|
|
|
|
AffixTokenMatcherWarehouse* fTokenWarehouse;
|
2018-02-10 14:29:26 +00:00
|
|
|
|
2018-02-10 15:49:02 +00:00
|
|
|
friend class AffixMatcher;
|
|
|
|
|
2018-02-10 14:29:26 +00:00
|
|
|
static bool isInteresting(const AffixPatternProvider& patternInfo, const IgnorablesMatcher& ignorables,
|
|
|
|
parse_flags_t parseFlags, UErrorCode& status);
|
|
|
|
};
|
|
|
|
|
|
|
|
|
2018-02-10 06:36:07 +00:00
|
|
|
} // namespace impl
|
|
|
|
} // namespace numparse
|
|
|
|
U_NAMESPACE_END
|
|
|
|
|
|
|
|
#endif //__NUMPARSE_AFFIXES_H__
|
|
|
|
#endif /* #if !UCONFIG_NO_FORMATTING */
|