2017-05-12 09:59:18 +00:00
|
|
|
#!/usr/bin/env python2
|
2011-04-27 10:05:43 +00:00
|
|
|
#############################################################################
|
|
|
|
##
|
2017-05-30 13:50:47 +00:00
|
|
|
## Copyright (C) 2017 The Qt Company Ltd.
|
2016-01-15 12:36:27 +00:00
|
|
|
## Contact: https://www.qt.io/licensing/
|
2011-04-27 10:05:43 +00:00
|
|
|
##
|
|
|
|
## This file is part of the test suite of the Qt Toolkit.
|
|
|
|
##
|
2016-01-15 12:36:27 +00:00
|
|
|
## $QT_BEGIN_LICENSE:GPL-EXCEPT$
|
2012-09-19 12:28:29 +00:00
|
|
|
## Commercial License Usage
|
|
|
|
## Licensees holding valid commercial Qt licenses may use this file in
|
|
|
|
## accordance with the commercial license agreement provided with the
|
|
|
|
## Software or, alternatively, in accordance with the terms contained in
|
2015-01-28 08:44:43 +00:00
|
|
|
## a written agreement between you and The Qt Company. For licensing terms
|
2016-01-15 12:36:27 +00:00
|
|
|
## and conditions see https://www.qt.io/terms-conditions. For further
|
|
|
|
## information use the contact form at https://www.qt.io/contact-us.
|
2012-09-19 12:28:29 +00:00
|
|
|
##
|
2016-01-15 12:36:27 +00:00
|
|
|
## GNU General Public License Usage
|
|
|
|
## Alternatively, this file may be used under the terms of the GNU
|
|
|
|
## General Public License version 3 as published by the Free Software
|
|
|
|
## Foundation with exceptions as appearing in the file LICENSE.GPL3-EXCEPT
|
|
|
|
## included in the packaging of this file. Please review the following
|
|
|
|
## information to ensure the GNU General Public License requirements will
|
|
|
|
## be met: https://www.gnu.org/licenses/gpl-3.0.html.
|
2011-04-27 10:05:43 +00:00
|
|
|
##
|
|
|
|
## $QT_END_LICENSE$
|
|
|
|
##
|
|
|
|
#############################################################################
|
2017-05-23 13:24:35 +00:00
|
|
|
"""Convert CLDR data to qLocaleXML
|
|
|
|
|
|
|
|
The CLDR data can be downloaded from CLDR_, which has a sub-directory
|
|
|
|
for each version; you need the ``core.zip`` file for your version of
|
|
|
|
choice (typically the latest). This script has had updates to cope up
|
|
|
|
to v29; for later versions, we may need adaptations. Unpack the
|
|
|
|
downloaded ``core.zip`` and check it has a common/main/ sub-directory:
|
|
|
|
pass the path of that sub-directory to this script as its single
|
|
|
|
command-line argument. Save its standard output (but not error) to a
|
|
|
|
file for later processing by ``./qlocalexml2cpp.py``
|
|
|
|
|
|
|
|
.. _CLDR: ftp://unicode.org/Public/cldr/
|
|
|
|
"""
|
2011-04-27 10:05:43 +00:00
|
|
|
|
|
|
|
import os
|
|
|
|
import sys
|
2017-05-30 13:50:47 +00:00
|
|
|
import re
|
|
|
|
|
2011-04-27 10:05:43 +00:00
|
|
|
import enumdata
|
|
|
|
import xpathlite
|
2017-05-31 19:42:11 +00:00
|
|
|
from xpathlite import DraftResolution, findAlias, findEntry, findTagsInFile
|
2011-04-27 10:05:43 +00:00
|
|
|
from dateconverter import convert_date
|
2017-05-30 13:50:47 +00:00
|
|
|
from localexml import Locale
|
2011-04-27 10:05:43 +00:00
|
|
|
|
|
|
|
findEntryInFile = xpathlite._findEntryInFile
|
|
|
|
|
|
|
|
def parse_number_format(patterns, data):
|
|
|
|
# this is a very limited parsing of the number format for currency only.
|
|
|
|
def skip_repeating_pattern(x):
|
|
|
|
p = x.replace('0', '#').replace(',', '').replace('.', '')
|
|
|
|
seen = False
|
|
|
|
result = ''
|
|
|
|
for c in p:
|
|
|
|
if c == '#':
|
|
|
|
if seen:
|
|
|
|
continue
|
|
|
|
seen = True
|
|
|
|
else:
|
|
|
|
seen = False
|
|
|
|
result = result + c
|
|
|
|
return result
|
|
|
|
patterns = patterns.split(';')
|
|
|
|
result = []
|
|
|
|
for pattern in patterns:
|
|
|
|
pattern = skip_repeating_pattern(pattern)
|
|
|
|
pattern = pattern.replace('#', "%1")
|
|
|
|
# according to http://www.unicode.org/reports/tr35/#Number_Format_Patterns
|
|
|
|
# there can be doubled or trippled currency sign, however none of the
|
|
|
|
# locales use that.
|
|
|
|
pattern = pattern.replace(u'\xa4', "%2")
|
|
|
|
pattern = pattern.replace("''", "###").replace("'", '').replace("###", "'")
|
|
|
|
pattern = pattern.replace('-', data['minus'])
|
|
|
|
pattern = pattern.replace('+', data['plus'])
|
|
|
|
result.append(pattern)
|
|
|
|
return result
|
|
|
|
|
|
|
|
def parse_list_pattern_part_format(pattern):
|
2017-05-31 19:42:11 +00:00
|
|
|
# This is a very limited parsing of the format for list pattern part only.
|
|
|
|
return pattern.replace("{0}", "%1").replace("{1}", "%2").replace("{2}", "%3")
|
2011-04-27 10:05:43 +00:00
|
|
|
|
|
|
|
def generateLocaleInfo(path):
|
|
|
|
if not path.endswith(".xml"):
|
|
|
|
return {}
|
2012-11-21 04:08:24 +00:00
|
|
|
|
|
|
|
# skip legacy/compatibility ones
|
|
|
|
alias = findAlias(path)
|
|
|
|
if alias:
|
2017-05-12 09:59:18 +00:00
|
|
|
raise xpathlite.Error('alias to "%s"' % alias)
|
2012-11-21 04:08:24 +00:00
|
|
|
|
2017-05-31 19:42:11 +00:00
|
|
|
def code(tag):
|
|
|
|
return findEntryInFile(path, 'identity/' + tag, attribute="type")[0]
|
2011-04-27 10:05:43 +00:00
|
|
|
|
2017-05-31 19:42:11 +00:00
|
|
|
return _generateLocaleInfo(path, code('language'), code('script'),
|
|
|
|
code('territory'), code('variant'))
|
[QLocaleData] Extract defaultContent locales
This adds some locales missing in the common/main/ directory, namely:
bss_CM, cch_NG, dv_MV, gaa_GH, gez_ET, ha_Arab_NG, iu_Cans_CA, kaj_NG,
kcg_NG, kpe_LR, ku_Latn_TR, mi_NZ, ms_Arab_MY, mn_Mong_CN, nds_DE,
ny_MW, oc_FR, sa_IN, sid_ET, tk_Latn_TM, trv_TW, tt_RU, ug_Arab_CN,
wa_BE, wo_Latn_SN
See http://www.unicode.org/reports/tr35/tr35-info.html#Default_Content
for more info.
Change-Id: I6b3082d370a21da64fbd5e72ab6344e1d7a6a3c9
Reviewed-by: Lars Knoll <lars.knoll@digia.com>
2015-03-20 21:12:30 +00:00
|
|
|
|
|
|
|
def _generateLocaleInfo(path, language_code, script_code, country_code, variant_code=""):
|
|
|
|
if not path.endswith(".xml"):
|
|
|
|
return {}
|
|
|
|
|
|
|
|
if language_code == 'root':
|
|
|
|
# just skip it
|
|
|
|
return {}
|
|
|
|
|
2011-04-27 10:05:43 +00:00
|
|
|
# we do not support variants
|
|
|
|
# ### actually there is only one locale with variant: en_US_POSIX
|
|
|
|
# does anybody care about it at all?
|
|
|
|
if variant_code:
|
2017-05-12 09:59:18 +00:00
|
|
|
raise xpathlite.Error('we do not support variants ("%s")' % variant_code)
|
2011-04-27 10:05:43 +00:00
|
|
|
|
|
|
|
language_id = enumdata.languageCodeToId(language_code)
|
2012-11-14 16:09:02 +00:00
|
|
|
if language_id <= 0:
|
2017-05-12 09:59:18 +00:00
|
|
|
raise xpathlite.Error('unknown language code "%s"' % language_code)
|
2011-04-27 10:05:43 +00:00
|
|
|
|
|
|
|
script_id = enumdata.scriptCodeToId(script_code)
|
2012-11-14 16:09:02 +00:00
|
|
|
if script_id == -1:
|
2017-05-12 09:59:18 +00:00
|
|
|
raise xpathlite.Error('unknown script code "%s"' % script_code)
|
2011-04-27 10:05:43 +00:00
|
|
|
|
2012-11-21 13:45:18 +00:00
|
|
|
# we should handle fully qualified names with the territory
|
|
|
|
if not country_code:
|
|
|
|
return {}
|
2011-04-27 10:05:43 +00:00
|
|
|
country_id = enumdata.countryCodeToId(country_code)
|
2012-11-14 16:09:02 +00:00
|
|
|
if country_id <= 0:
|
2017-05-12 09:59:18 +00:00
|
|
|
raise xpathlite.Error('unknown country code "%s"' % country_code)
|
2011-04-27 10:05:43 +00:00
|
|
|
|
|
|
|
# So we say we accept only those values that have "contributed" or
|
|
|
|
# "approved" resolution. see http://www.unicode.org/cldr/process.html
|
|
|
|
# But we only respect the resolution for new datas for backward
|
|
|
|
# compatibility.
|
|
|
|
draft = DraftResolution.contributed
|
|
|
|
|
2017-05-31 19:42:11 +00:00
|
|
|
result = dict(
|
|
|
|
language=enumdata.language_list[language_id][0],
|
|
|
|
language_code=language_code, language_id=language_id,
|
|
|
|
script=enumdata.script_list[script_id][0],
|
|
|
|
script_code=script_code, script_id=script_id,
|
|
|
|
country=enumdata.country_list[country_id][0],
|
|
|
|
country_code=country_code, country_id=country_id,
|
|
|
|
variant_code=variant_code)
|
2011-04-27 10:05:43 +00:00
|
|
|
|
[QLocaleData] Extract defaultContent locales
This adds some locales missing in the common/main/ directory, namely:
bss_CM, cch_NG, dv_MV, gaa_GH, gez_ET, ha_Arab_NG, iu_Cans_CA, kaj_NG,
kcg_NG, kpe_LR, ku_Latn_TR, mi_NZ, ms_Arab_MY, mn_Mong_CN, nds_DE,
ny_MW, oc_FR, sa_IN, sid_ET, tk_Latn_TM, trv_TW, tt_RU, ug_Arab_CN,
wa_BE, wo_Latn_SN
See http://www.unicode.org/reports/tr35/tr35-info.html#Default_Content
for more info.
Change-Id: I6b3082d370a21da64fbd5e72ab6344e1d7a6a3c9
Reviewed-by: Lars Knoll <lars.knoll@digia.com>
2015-03-20 21:12:30 +00:00
|
|
|
(dir_name, file_name) = os.path.split(path)
|
2017-05-31 19:42:11 +00:00
|
|
|
def from_supplement(tag,
|
|
|
|
path=os.path.join(dir_name, '..', 'supplemental',
|
|
|
|
'supplementalData.xml')):
|
|
|
|
return findTagsInFile(path, tag)
|
|
|
|
currencies = from_supplement('currencyData/region[iso3166=%s]' % country_code)
|
2011-04-27 10:05:43 +00:00
|
|
|
result['currencyIsoCode'] = ''
|
|
|
|
result['currencyDigits'] = 2
|
|
|
|
result['currencyRounding'] = 1
|
|
|
|
if currencies:
|
|
|
|
for e in currencies:
|
|
|
|
if e[0] == 'currency':
|
2017-05-31 19:42:11 +00:00
|
|
|
t = [x[1] == 'false' for x in e[1] if x[0] == 'tender']
|
|
|
|
if t and t[0]:
|
|
|
|
pass
|
|
|
|
elif not any(x[0] == 'to' for x in e[1]):
|
2017-05-12 09:59:18 +00:00
|
|
|
result['currencyIsoCode'] = (x[1] for x in e[1] if x[0] == 'iso4217').next()
|
2011-04-27 10:05:43 +00:00
|
|
|
break
|
|
|
|
if result['currencyIsoCode']:
|
2017-05-31 19:42:11 +00:00
|
|
|
t = from_supplement("currencyData/fractions/info[iso4217=%s]"
|
|
|
|
% result['currencyIsoCode'])
|
2011-04-27 10:05:43 +00:00
|
|
|
if t and t[0][0] == 'info':
|
2017-05-12 09:59:18 +00:00
|
|
|
result['currencyDigits'] = (int(x[1]) for x in t[0][1] if x[0] == 'digits').next()
|
|
|
|
result['currencyRounding'] = (int(x[1]) for x in t[0][1] if x[0] == 'rounding').next()
|
2011-04-27 10:05:43 +00:00
|
|
|
numbering_system = None
|
|
|
|
try:
|
|
|
|
numbering_system = findEntry(path, "numbers/defaultNumberingSystem")
|
|
|
|
except:
|
|
|
|
pass
|
|
|
|
def findEntryDef(path, xpath, value=''):
|
|
|
|
try:
|
|
|
|
return findEntry(path, xpath)
|
|
|
|
except xpathlite.Error:
|
|
|
|
return value
|
|
|
|
def get_number_in_system(path, xpath, numbering_system):
|
|
|
|
if numbering_system:
|
|
|
|
try:
|
|
|
|
return findEntry(path, xpath + "[numberSystem=" + numbering_system + "]")
|
|
|
|
except xpathlite.Error:
|
2012-03-22 12:07:07 +00:00
|
|
|
# in CLDR 1.9 number system was refactored for numbers (but not for currency)
|
|
|
|
# so if previous findEntry doesn't work we should try this:
|
|
|
|
try:
|
|
|
|
return findEntry(path, xpath.replace("/symbols/", "/symbols[numberSystem=" + numbering_system + "]/"))
|
|
|
|
except xpathlite.Error:
|
|
|
|
# fallback to default
|
|
|
|
pass
|
2011-04-27 10:05:43 +00:00
|
|
|
return findEntry(path, xpath)
|
2012-03-22 12:07:07 +00:00
|
|
|
|
2011-04-27 10:05:43 +00:00
|
|
|
result['decimal'] = get_number_in_system(path, "numbers/symbols/decimal", numbering_system)
|
|
|
|
result['group'] = get_number_in_system(path, "numbers/symbols/group", numbering_system)
|
|
|
|
result['list'] = get_number_in_system(path, "numbers/symbols/list", numbering_system)
|
|
|
|
result['percent'] = get_number_in_system(path, "numbers/symbols/percentSign", numbering_system)
|
2012-03-22 12:07:07 +00:00
|
|
|
try:
|
|
|
|
numbering_systems = {}
|
2017-05-31 19:42:11 +00:00
|
|
|
for ns in findTagsInFile(os.path.join(cldr_dir, '..', 'supplemental',
|
|
|
|
'numberingSystems.xml'),
|
|
|
|
'numberingSystems'):
|
2012-03-22 12:07:07 +00:00
|
|
|
tmp = {}
|
|
|
|
id = ""
|
|
|
|
for data in ns[1:][0]: # ns looks like this: [u'numberingSystem', [(u'digits', u'0123456789'), (u'type', u'numeric'), (u'id', u'latn')]]
|
|
|
|
tmp[data[0]] = data[1]
|
|
|
|
if data[0] == u"id":
|
|
|
|
id = data[1]
|
|
|
|
numbering_systems[id] = tmp
|
|
|
|
result['zero'] = numbering_systems[numbering_system][u"digits"][0]
|
|
|
|
except e:
|
|
|
|
sys.stderr.write("Native zero detection problem:\n" + str(e) + "\n")
|
|
|
|
result['zero'] = get_number_in_system(path, "numbers/symbols/nativeZeroDigit", numbering_system)
|
2011-04-27 10:05:43 +00:00
|
|
|
result['minus'] = get_number_in_system(path, "numbers/symbols/minusSign", numbering_system)
|
|
|
|
result['plus'] = get_number_in_system(path, "numbers/symbols/plusSign", numbering_system)
|
|
|
|
result['exp'] = get_number_in_system(path, "numbers/symbols/exponential", numbering_system).lower()
|
|
|
|
result['quotationStart'] = findEntry(path, "delimiters/quotationStart")
|
|
|
|
result['quotationEnd'] = findEntry(path, "delimiters/quotationEnd")
|
|
|
|
result['alternateQuotationStart'] = findEntry(path, "delimiters/alternateQuotationStart")
|
|
|
|
result['alternateQuotationEnd'] = findEntry(path, "delimiters/alternateQuotationEnd")
|
|
|
|
result['listPatternPartStart'] = parse_list_pattern_part_format(findEntry(path, "listPatterns/listPattern/listPatternPart[start]"))
|
|
|
|
result['listPatternPartMiddle'] = parse_list_pattern_part_format(findEntry(path, "listPatterns/listPattern/listPatternPart[middle]"))
|
|
|
|
result['listPatternPartEnd'] = parse_list_pattern_part_format(findEntry(path, "listPatterns/listPattern/listPatternPart[end]"))
|
|
|
|
result['listPatternPartTwo'] = parse_list_pattern_part_format(findEntry(path, "listPatterns/listPattern/listPatternPart[2]"))
|
|
|
|
result['am'] = findEntry(path, "dates/calendars/calendar[gregorian]/dayPeriods/dayPeriodContext[format]/dayPeriodWidth[wide]/dayPeriod[am]", draft)
|
|
|
|
result['pm'] = findEntry(path, "dates/calendars/calendar[gregorian]/dayPeriods/dayPeriodContext[format]/dayPeriodWidth[wide]/dayPeriod[pm]", draft)
|
|
|
|
result['longDateFormat'] = convert_date(findEntry(path, "dates/calendars/calendar[gregorian]/dateFormats/dateFormatLength[full]/dateFormat/pattern"))
|
|
|
|
result['shortDateFormat'] = convert_date(findEntry(path, "dates/calendars/calendar[gregorian]/dateFormats/dateFormatLength[short]/dateFormat/pattern"))
|
|
|
|
result['longTimeFormat'] = convert_date(findEntry(path, "dates/calendars/calendar[gregorian]/timeFormats/timeFormatLength[full]/timeFormat/pattern"))
|
|
|
|
result['shortTimeFormat'] = convert_date(findEntry(path, "dates/calendars/calendar[gregorian]/timeFormats/timeFormatLength[short]/timeFormat/pattern"))
|
|
|
|
|
|
|
|
endonym = None
|
|
|
|
if country_code and script_code:
|
|
|
|
endonym = findEntryDef(path, "localeDisplayNames/languages/language[type=%s_%s_%s]" % (language_code, script_code, country_code))
|
|
|
|
if not endonym and script_code:
|
|
|
|
endonym = findEntryDef(path, "localeDisplayNames/languages/language[type=%s_%s]" % (language_code, script_code))
|
|
|
|
if not endonym and country_code:
|
|
|
|
endonym = findEntryDef(path, "localeDisplayNames/languages/language[type=%s_%s]" % (language_code, country_code))
|
|
|
|
if not endonym:
|
|
|
|
endonym = findEntryDef(path, "localeDisplayNames/languages/language[type=%s]" % (language_code))
|
|
|
|
result['language_endonym'] = endonym
|
|
|
|
result['country_endonym'] = findEntryDef(path, "localeDisplayNames/territories/territory[type=%s]" % (country_code))
|
|
|
|
|
|
|
|
currency_format = get_number_in_system(path, "numbers/currencyFormats/currencyFormatLength/currencyFormat/pattern", numbering_system)
|
|
|
|
currency_format = parse_number_format(currency_format, result)
|
|
|
|
result['currencyFormat'] = currency_format[0]
|
|
|
|
result['currencyNegativeFormat'] = ''
|
|
|
|
if len(currency_format) > 1:
|
|
|
|
result['currencyNegativeFormat'] = currency_format[1]
|
|
|
|
|
|
|
|
result['currencySymbol'] = ''
|
|
|
|
result['currencyDisplayName'] = ''
|
|
|
|
if result['currencyIsoCode']:
|
|
|
|
result['currencySymbol'] = findEntryDef(path, "numbers/currencies/currency[%s]/symbol" % result['currencyIsoCode'])
|
2017-05-12 09:59:18 +00:00
|
|
|
result['currencyDisplayName'] = ';'.join(
|
|
|
|
findEntryDef(path, 'numbers/currencies/currency[' + result['currencyIsoCode']
|
|
|
|
+ ']/displayName' + tail)
|
|
|
|
for tail in ['',] + [
|
|
|
|
'[count=%s]' % x for x in ('zero', 'one', 'two', 'few', 'many', 'other')
|
|
|
|
]) + ';'
|
|
|
|
|
|
|
|
# Used for month and day data:
|
|
|
|
namings = (
|
|
|
|
('standaloneLong', 'stand-alone', 'wide'),
|
|
|
|
('standaloneShort', 'stand-alone', 'abbreviated'),
|
|
|
|
('standaloneNarrow', 'stand-alone', 'narrow'),
|
|
|
|
('long', 'format', 'wide'),
|
|
|
|
('short', 'format', 'abbreviated'),
|
|
|
|
('narrow', 'format', 'narrow'),
|
|
|
|
)
|
|
|
|
|
|
|
|
# Month data:
|
|
|
|
for cal in ('gregorian',): # We shall want to add to this
|
|
|
|
stem = 'dates/calendars/calendar[' + cal + ']/months/'
|
|
|
|
for (key, mode, size) in namings:
|
|
|
|
prop = 'monthContext[' + mode + ']/monthWidth[' + size + ']/'
|
|
|
|
result[key + 'Months'] = ';'.join(
|
|
|
|
findEntry(path, stem + prop + "month[%d]" % i)
|
|
|
|
for i in range(1, 13)) + ';'
|
|
|
|
|
|
|
|
# Day data (for Gregorian, at least):
|
|
|
|
stem = 'dates/calendars/calendar[gregorian]/days/'
|
|
|
|
days = ('sun', 'mon', 'tue', 'wed', 'thu', 'fri', 'sat')
|
|
|
|
for (key, mode, size) in namings:
|
|
|
|
prop = 'dayContext[' + mode + ']/dayWidth[' + size + ']/day'
|
|
|
|
result[key + 'Days'] = ';'.join(
|
|
|
|
findEntry(path, stem + prop + '[' + day + ']')
|
|
|
|
for day in days) + ';'
|
2011-04-27 10:05:43 +00:00
|
|
|
|
2017-05-30 13:50:47 +00:00
|
|
|
return Locale(result)
|
2011-04-27 10:05:43 +00:00
|
|
|
|
|
|
|
def addEscapes(s):
|
|
|
|
result = ''
|
|
|
|
for c in s:
|
|
|
|
n = ord(c)
|
|
|
|
if n < 128:
|
|
|
|
result += c
|
|
|
|
else:
|
|
|
|
result += "\\x"
|
|
|
|
result += "%02x" % (n)
|
|
|
|
return result
|
|
|
|
|
|
|
|
def unicodeStr(s):
|
|
|
|
utf8 = s.encode('utf-8')
|
|
|
|
return "<size>" + str(len(utf8)) + "</size><data>" + addEscapes(utf8) + "</data>"
|
|
|
|
|
|
|
|
def usage():
|
|
|
|
print "Usage: cldr2qlocalexml.py <path-to-cldr-main>"
|
|
|
|
sys.exit()
|
|
|
|
|
|
|
|
def integrateWeekData(filePath):
|
|
|
|
if not filePath.endswith(".xml"):
|
|
|
|
return {}
|
2017-05-31 17:34:09 +00:00
|
|
|
|
|
|
|
def lookup(key):
|
|
|
|
return findEntryInFile(filePath, key, attribute='territories')[0].split()
|
|
|
|
days = ('mon', 'tue', 'wed', 'thu', 'fri', 'sat', 'sun')
|
2011-04-27 10:05:43 +00:00
|
|
|
|
|
|
|
firstDayByCountryCode = {}
|
2017-05-31 17:34:09 +00:00
|
|
|
for day in days:
|
|
|
|
for countryCode in lookup('weekData/firstDay[day=%s]' % day):
|
|
|
|
firstDayByCountryCode[countryCode] = day
|
2011-04-27 10:05:43 +00:00
|
|
|
|
|
|
|
weekendStartByCountryCode = {}
|
2017-05-31 17:34:09 +00:00
|
|
|
for day in days:
|
|
|
|
for countryCode in lookup('weekData/weekendStart[day=%s]' % day):
|
|
|
|
weekendStartByCountryCode[countryCode] = day
|
2011-04-27 10:05:43 +00:00
|
|
|
|
|
|
|
weekendEndByCountryCode = {}
|
2017-05-31 17:34:09 +00:00
|
|
|
for day in days:
|
|
|
|
for countryCode in lookup('weekData/weekendEnd[day=%s]' % day):
|
|
|
|
weekendEndByCountryCode[countryCode] = day
|
2011-04-27 10:05:43 +00:00
|
|
|
|
2017-05-30 13:50:47 +00:00
|
|
|
for (key, locale) in locale_database.iteritems():
|
|
|
|
countryCode = locale.country_code
|
2011-04-27 10:05:43 +00:00
|
|
|
if countryCode in firstDayByCountryCode:
|
2017-05-30 13:50:47 +00:00
|
|
|
locale.firstDayOfWeek = firstDayByCountryCode[countryCode]
|
2011-04-27 10:05:43 +00:00
|
|
|
else:
|
2017-05-30 13:50:47 +00:00
|
|
|
locale.firstDayOfWeek = firstDayByCountryCode["001"]
|
2011-04-27 10:05:43 +00:00
|
|
|
|
|
|
|
if countryCode in weekendStartByCountryCode:
|
2017-05-30 13:50:47 +00:00
|
|
|
locale.weekendStart = weekendStartByCountryCode[countryCode]
|
2011-04-27 10:05:43 +00:00
|
|
|
else:
|
2017-05-30 13:50:47 +00:00
|
|
|
locale.weekendStart = weekendStartByCountryCode["001"]
|
2011-04-27 10:05:43 +00:00
|
|
|
|
|
|
|
if countryCode in weekendEndByCountryCode:
|
2017-05-30 13:50:47 +00:00
|
|
|
locale.weekendEnd = weekendEndByCountryCode[countryCode]
|
2011-04-27 10:05:43 +00:00
|
|
|
else:
|
2017-05-30 13:50:47 +00:00
|
|
|
locale.weekendEnd = weekendEndByCountryCode["001"]
|
2011-04-27 10:05:43 +00:00
|
|
|
|
|
|
|
if len(sys.argv) != 2:
|
|
|
|
usage()
|
|
|
|
|
|
|
|
cldr_dir = sys.argv[1]
|
|
|
|
|
|
|
|
if not os.path.isdir(cldr_dir):
|
|
|
|
usage()
|
|
|
|
|
|
|
|
cldr_files = os.listdir(cldr_dir)
|
|
|
|
|
|
|
|
locale_database = {}
|
[QLocaleData] Extract defaultContent locales
This adds some locales missing in the common/main/ directory, namely:
bss_CM, cch_NG, dv_MV, gaa_GH, gez_ET, ha_Arab_NG, iu_Cans_CA, kaj_NG,
kcg_NG, kpe_LR, ku_Latn_TR, mi_NZ, ms_Arab_MY, mn_Mong_CN, nds_DE,
ny_MW, oc_FR, sa_IN, sid_ET, tk_Latn_TM, trv_TW, tt_RU, ug_Arab_CN,
wa_BE, wo_Latn_SN
See http://www.unicode.org/reports/tr35/tr35-info.html#Default_Content
for more info.
Change-Id: I6b3082d370a21da64fbd5e72ab6344e1d7a6a3c9
Reviewed-by: Lars Knoll <lars.knoll@digia.com>
2015-03-20 21:12:30 +00:00
|
|
|
|
|
|
|
# see http://www.unicode.org/reports/tr35/tr35-info.html#Default_Content
|
|
|
|
defaultContent_locales = {}
|
2017-05-31 19:42:11 +00:00
|
|
|
for ns in findTagsInFile(os.path.join(cldr_dir, '..', 'supplemental',
|
|
|
|
'supplementalMetadata.xml'),
|
|
|
|
'metadata/defaultContent'):
|
[QLocaleData] Extract defaultContent locales
This adds some locales missing in the common/main/ directory, namely:
bss_CM, cch_NG, dv_MV, gaa_GH, gez_ET, ha_Arab_NG, iu_Cans_CA, kaj_NG,
kcg_NG, kpe_LR, ku_Latn_TR, mi_NZ, ms_Arab_MY, mn_Mong_CN, nds_DE,
ny_MW, oc_FR, sa_IN, sid_ET, tk_Latn_TM, trv_TW, tt_RU, ug_Arab_CN,
wa_BE, wo_Latn_SN
See http://www.unicode.org/reports/tr35/tr35-info.html#Default_Content
for more info.
Change-Id: I6b3082d370a21da64fbd5e72ab6344e1d7a6a3c9
Reviewed-by: Lars Knoll <lars.knoll@digia.com>
2015-03-20 21:12:30 +00:00
|
|
|
for data in ns[1:][0]:
|
|
|
|
if data[0] == u"locales":
|
|
|
|
defaultContent_locales = data[1].split()
|
|
|
|
|
|
|
|
for file in defaultContent_locales:
|
|
|
|
items = file.split("_")
|
|
|
|
if len(items) == 3:
|
|
|
|
language_code = items[0]
|
|
|
|
script_code = items[1]
|
|
|
|
country_code = items[2]
|
|
|
|
else:
|
|
|
|
if len(items) != 2:
|
2017-05-23 13:57:40 +00:00
|
|
|
sys.stderr.write('skipping defaultContent locale "' + file + '" [neither lang_script_country nor lang_country]\n')
|
[QLocaleData] Extract defaultContent locales
This adds some locales missing in the common/main/ directory, namely:
bss_CM, cch_NG, dv_MV, gaa_GH, gez_ET, ha_Arab_NG, iu_Cans_CA, kaj_NG,
kcg_NG, kpe_LR, ku_Latn_TR, mi_NZ, ms_Arab_MY, mn_Mong_CN, nds_DE,
ny_MW, oc_FR, sa_IN, sid_ET, tk_Latn_TM, trv_TW, tt_RU, ug_Arab_CN,
wa_BE, wo_Latn_SN
See http://www.unicode.org/reports/tr35/tr35-info.html#Default_Content
for more info.
Change-Id: I6b3082d370a21da64fbd5e72ab6344e1d7a6a3c9
Reviewed-by: Lars Knoll <lars.knoll@digia.com>
2015-03-20 21:12:30 +00:00
|
|
|
continue
|
|
|
|
language_code = items[0]
|
|
|
|
script_code = ""
|
|
|
|
country_code = items[1]
|
|
|
|
if len(country_code) == 4:
|
2017-05-23 13:57:40 +00:00
|
|
|
sys.stderr.write('skipping defaultContent locale "' + file + '" [long country code]\n')
|
[QLocaleData] Extract defaultContent locales
This adds some locales missing in the common/main/ directory, namely:
bss_CM, cch_NG, dv_MV, gaa_GH, gez_ET, ha_Arab_NG, iu_Cans_CA, kaj_NG,
kcg_NG, kpe_LR, ku_Latn_TR, mi_NZ, ms_Arab_MY, mn_Mong_CN, nds_DE,
ny_MW, oc_FR, sa_IN, sid_ET, tk_Latn_TM, trv_TW, tt_RU, ug_Arab_CN,
wa_BE, wo_Latn_SN
See http://www.unicode.org/reports/tr35/tr35-info.html#Default_Content
for more info.
Change-Id: I6b3082d370a21da64fbd5e72ab6344e1d7a6a3c9
Reviewed-by: Lars Knoll <lars.knoll@digia.com>
2015-03-20 21:12:30 +00:00
|
|
|
continue
|
|
|
|
try:
|
|
|
|
l = _generateLocaleInfo(cldr_dir + "/" + file + ".xml", language_code, script_code, country_code)
|
|
|
|
if not l:
|
2017-05-23 13:57:40 +00:00
|
|
|
sys.stderr.write('skipping defaultContent locale "' + file + '" [no locale info generated]\n')
|
[QLocaleData] Extract defaultContent locales
This adds some locales missing in the common/main/ directory, namely:
bss_CM, cch_NG, dv_MV, gaa_GH, gez_ET, ha_Arab_NG, iu_Cans_CA, kaj_NG,
kcg_NG, kpe_LR, ku_Latn_TR, mi_NZ, ms_Arab_MY, mn_Mong_CN, nds_DE,
ny_MW, oc_FR, sa_IN, sid_ET, tk_Latn_TM, trv_TW, tt_RU, ug_Arab_CN,
wa_BE, wo_Latn_SN
See http://www.unicode.org/reports/tr35/tr35-info.html#Default_Content
for more info.
Change-Id: I6b3082d370a21da64fbd5e72ab6344e1d7a6a3c9
Reviewed-by: Lars Knoll <lars.knoll@digia.com>
2015-03-20 21:12:30 +00:00
|
|
|
continue
|
|
|
|
except xpathlite.Error as e:
|
2017-05-12 09:59:18 +00:00
|
|
|
sys.stderr.write('skipping defaultContent locale "%s" (%s)\n' % (file, str(e)))
|
[QLocaleData] Extract defaultContent locales
This adds some locales missing in the common/main/ directory, namely:
bss_CM, cch_NG, dv_MV, gaa_GH, gez_ET, ha_Arab_NG, iu_Cans_CA, kaj_NG,
kcg_NG, kpe_LR, ku_Latn_TR, mi_NZ, ms_Arab_MY, mn_Mong_CN, nds_DE,
ny_MW, oc_FR, sa_IN, sid_ET, tk_Latn_TM, trv_TW, tt_RU, ug_Arab_CN,
wa_BE, wo_Latn_SN
See http://www.unicode.org/reports/tr35/tr35-info.html#Default_Content
for more info.
Change-Id: I6b3082d370a21da64fbd5e72ab6344e1d7a6a3c9
Reviewed-by: Lars Knoll <lars.knoll@digia.com>
2015-03-20 21:12:30 +00:00
|
|
|
continue
|
|
|
|
|
2017-05-30 13:50:47 +00:00
|
|
|
locale_database[(l.language_id, l.script_id, l.country_id, l.variant_code)] = l
|
[QLocaleData] Extract defaultContent locales
This adds some locales missing in the common/main/ directory, namely:
bss_CM, cch_NG, dv_MV, gaa_GH, gez_ET, ha_Arab_NG, iu_Cans_CA, kaj_NG,
kcg_NG, kpe_LR, ku_Latn_TR, mi_NZ, ms_Arab_MY, mn_Mong_CN, nds_DE,
ny_MW, oc_FR, sa_IN, sid_ET, tk_Latn_TM, trv_TW, tt_RU, ug_Arab_CN,
wa_BE, wo_Latn_SN
See http://www.unicode.org/reports/tr35/tr35-info.html#Default_Content
for more info.
Change-Id: I6b3082d370a21da64fbd5e72ab6344e1d7a6a3c9
Reviewed-by: Lars Knoll <lars.knoll@digia.com>
2015-03-20 21:12:30 +00:00
|
|
|
|
2011-04-27 10:05:43 +00:00
|
|
|
for file in cldr_files:
|
2012-11-21 04:08:24 +00:00
|
|
|
try:
|
|
|
|
l = generateLocaleInfo(cldr_dir + "/" + file)
|
|
|
|
if not l:
|
2017-05-23 13:57:40 +00:00
|
|
|
sys.stderr.write('skipping file "' + file + '" [no locale info generated]\n')
|
2012-11-21 04:08:24 +00:00
|
|
|
continue
|
|
|
|
except xpathlite.Error as e:
|
2017-05-12 09:59:18 +00:00
|
|
|
sys.stderr.write('skipping file "%s" (%s)\n' % (file, str(e)))
|
2011-04-27 10:05:43 +00:00
|
|
|
continue
|
|
|
|
|
2017-05-30 13:50:47 +00:00
|
|
|
locale_database[(l.language_id, l.script_id, l.country_id, l.variant_code)] = l
|
2011-04-27 10:05:43 +00:00
|
|
|
|
|
|
|
integrateWeekData(cldr_dir+"/../supplemental/supplementalData.xml")
|
|
|
|
locale_keys = locale_database.keys()
|
|
|
|
locale_keys.sort()
|
|
|
|
|
|
|
|
cldr_version = 'unknown'
|
|
|
|
ldml = open(cldr_dir+"/../dtd/ldml.dtd", "r")
|
|
|
|
for line in ldml:
|
|
|
|
if 'version cldrVersion CDATA #FIXED' in line:
|
|
|
|
cldr_version = line.split('"')[1]
|
|
|
|
|
|
|
|
print "<localeDatabase>"
|
|
|
|
print " <version>" + cldr_version + "</version>"
|
|
|
|
print " <languageList>"
|
|
|
|
for id in enumdata.language_list:
|
|
|
|
l = enumdata.language_list[id]
|
|
|
|
print " <language>"
|
|
|
|
print " <name>" + l[0] + "</name>"
|
|
|
|
print " <id>" + str(id) + "</id>"
|
|
|
|
print " <code>" + l[1] + "</code>"
|
|
|
|
print " </language>"
|
|
|
|
print " </languageList>"
|
|
|
|
|
|
|
|
print " <scriptList>"
|
|
|
|
for id in enumdata.script_list:
|
|
|
|
l = enumdata.script_list[id]
|
|
|
|
print " <script>"
|
|
|
|
print " <name>" + l[0] + "</name>"
|
|
|
|
print " <id>" + str(id) + "</id>"
|
|
|
|
print " <code>" + l[1] + "</code>"
|
|
|
|
print " </script>"
|
|
|
|
print " </scriptList>"
|
|
|
|
|
|
|
|
print " <countryList>"
|
|
|
|
for id in enumdata.country_list:
|
|
|
|
l = enumdata.country_list[id]
|
|
|
|
print " <country>"
|
|
|
|
print " <name>" + l[0] + "</name>"
|
|
|
|
print " <id>" + str(id) + "</id>"
|
|
|
|
print " <code>" + l[1] + "</code>"
|
|
|
|
print " </country>"
|
|
|
|
print " </countryList>"
|
|
|
|
|
2012-11-19 17:12:58 +00:00
|
|
|
def _parseLocale(l):
|
|
|
|
language = "AnyLanguage"
|
|
|
|
script = "AnyScript"
|
|
|
|
country = "AnyCountry"
|
|
|
|
|
2012-11-21 04:08:24 +00:00
|
|
|
if l == "und":
|
|
|
|
raise xpathlite.Error("we are treating unknown locale like C")
|
2012-11-19 17:12:58 +00:00
|
|
|
|
|
|
|
items = l.split("_")
|
|
|
|
language_code = items[0]
|
|
|
|
if language_code != "und":
|
|
|
|
language_id = enumdata.languageCodeToId(language_code)
|
|
|
|
if language_id == -1:
|
2017-05-12 09:59:18 +00:00
|
|
|
raise xpathlite.Error('unknown language code "%s"' % language_code)
|
2012-11-19 17:12:58 +00:00
|
|
|
language = enumdata.language_list[language_id][0]
|
|
|
|
|
|
|
|
if len(items) > 1:
|
|
|
|
script_code = items[1]
|
|
|
|
country_code = ""
|
|
|
|
if len(items) > 2:
|
|
|
|
country_code = items[2]
|
|
|
|
if len(script_code) == 4:
|
|
|
|
script_id = enumdata.scriptCodeToId(script_code)
|
|
|
|
if script_id == -1:
|
2017-05-12 09:59:18 +00:00
|
|
|
raise xpathlite.Error('unknown script code "%s"' % script_code)
|
2012-11-19 17:12:58 +00:00
|
|
|
script = enumdata.script_list[script_id][0]
|
|
|
|
else:
|
|
|
|
country_code = script_code
|
|
|
|
if country_code:
|
|
|
|
country_id = enumdata.countryCodeToId(country_code)
|
|
|
|
if country_id == -1:
|
2017-05-12 09:59:18 +00:00
|
|
|
raise xpathlite.Error('unknown country code "%s"' % country_code)
|
2012-11-19 17:12:58 +00:00
|
|
|
country = enumdata.country_list[country_id][0]
|
|
|
|
|
|
|
|
return (language, script, country)
|
|
|
|
|
|
|
|
print " <likelySubtags>"
|
|
|
|
for ns in findTagsInFile(cldr_dir + "/../supplemental/likelySubtags.xml", "likelySubtags"):
|
|
|
|
tmp = {}
|
|
|
|
for data in ns[1:][0]: # ns looks like this: [u'likelySubtag', [(u'from', u'aa'), (u'to', u'aa_Latn_ET')]]
|
|
|
|
tmp[data[0]] = data[1]
|
|
|
|
|
2012-11-21 04:08:24 +00:00
|
|
|
try:
|
|
|
|
(from_language, from_script, from_country) = _parseLocale(tmp[u"from"])
|
|
|
|
except xpathlite.Error as e:
|
2017-05-12 09:59:18 +00:00
|
|
|
sys.stderr.write('skipping likelySubtag "%s" -> "%s" (%s)\n' % (tmp[u"from"], tmp[u"to"], str(e)))
|
2012-11-19 17:12:58 +00:00
|
|
|
continue
|
2012-11-21 04:08:24 +00:00
|
|
|
try:
|
|
|
|
(to_language, to_script, to_country) = _parseLocale(tmp[u"to"])
|
|
|
|
except xpathlite.Error as e:
|
2017-05-12 09:59:18 +00:00
|
|
|
sys.stderr.write('skipping likelySubtag "%s" -> "%s" (%s)\n' % (tmp[u"from"], tmp[u"to"], str(e)))
|
2012-11-19 17:12:58 +00:00
|
|
|
continue
|
|
|
|
# substitute according to http://www.unicode.org/reports/tr35/#Likely_Subtags
|
|
|
|
if to_country == "AnyCountry" and from_country != to_country:
|
|
|
|
to_country = from_country
|
|
|
|
if to_script == "AnyScript" and from_script != to_script:
|
|
|
|
to_script = from_script
|
|
|
|
|
|
|
|
print " <likelySubtag>"
|
|
|
|
print " <from>"
|
|
|
|
print " <language>" + from_language + "</language>"
|
|
|
|
print " <script>" + from_script + "</script>"
|
|
|
|
print " <country>" + from_country + "</country>"
|
|
|
|
print " </from>"
|
|
|
|
print " <to>"
|
|
|
|
print " <language>" + to_language + "</language>"
|
|
|
|
print " <script>" + to_script + "</script>"
|
|
|
|
print " <country>" + to_country + "</country>"
|
|
|
|
print " </to>"
|
|
|
|
print " </likelySubtag>"
|
|
|
|
print " </likelySubtags>"
|
2011-04-27 10:05:43 +00:00
|
|
|
|
|
|
|
print " <localeList>"
|
|
|
|
|
2017-05-30 13:50:47 +00:00
|
|
|
Locale.C().toXml()
|
2011-04-27 10:05:43 +00:00
|
|
|
for key in locale_keys:
|
2017-05-30 13:50:47 +00:00
|
|
|
locale_database[key].toXml()
|
|
|
|
|
2011-04-27 10:05:43 +00:00
|
|
|
print " </localeList>"
|
|
|
|
print "</localeDatabase>"
|