twitter_cldr 3.2.1 → 3.3.0
Sign up to get free protection for your applications and to get access to all the features.
- checksums.yaml +4 -4
- data/Gemfile +9 -4
- data/History.txt +11 -0
- data/README.md +40 -3
- data/Rakefile +26 -21
- data/lib/twitter_cldr/collation/collator.rb +3 -3
- data/lib/twitter_cldr/collation/sort_key_builder.rb +10 -10
- data/lib/twitter_cldr/data_readers/calendar_data_reader.rb +8 -8
- data/lib/twitter_cldr/data_readers/date_time_data_reader.rb +2 -2
- data/lib/twitter_cldr/data_readers/number_data_reader.rb +10 -10
- data/lib/twitter_cldr/data_readers/timespan_data_reader.rb +27 -27
- data/lib/twitter_cldr/formatters/list_formatter.rb +1 -1
- data/lib/twitter_cldr/formatters/numbers/currency_formatter.rb +3 -3
- data/lib/twitter_cldr/formatters/numbers/helpers/integer.rb +7 -5
- data/lib/twitter_cldr/formatters/numbers/rbnf/formatters.rb +1 -1
- data/lib/twitter_cldr/formatters/numbers/rbnf/rule.rb +1 -1
- data/lib/twitter_cldr/formatters/numbers/rbnf/rule_set.rb +3 -2
- data/lib/twitter_cldr/formatters/numbers/rbnf.rb +2 -2
- data/lib/twitter_cldr/formatters/plurals/rules.rb +1 -1
- data/lib/twitter_cldr/localized/localized_date.rb +2 -2
- data/lib/twitter_cldr/localized/localized_datetime.rb +8 -8
- data/lib/twitter_cldr/localized/localized_number.rb +7 -7
- data/lib/twitter_cldr/localized/localized_string.rb +33 -1
- data/lib/twitter_cldr/localized/localized_symbol.rb +9 -1
- data/lib/twitter_cldr/localized/localized_time.rb +2 -2
- data/lib/twitter_cldr/localized/localized_timespan.rb +10 -10
- data/lib/twitter_cldr/parsers/number_parser.rb +1 -1
- data/lib/twitter_cldr/parsers/parser.rb +5 -1
- data/lib/twitter_cldr/parsers/unicode_regex/character_class.rb +41 -2
- data/lib/twitter_cldr/parsers/unicode_regex/character_range.rb +8 -0
- data/lib/twitter_cldr/parsers/unicode_regex/character_set.rb +66 -23
- data/lib/twitter_cldr/parsers/unicode_regex/literal.rb +4 -0
- data/lib/twitter_cldr/parsers/unicode_regex/unicode_string.rb +6 -3
- data/lib/twitter_cldr/parsers/unicode_regex_parser.rb +65 -32
- data/lib/twitter_cldr/parsers.rb +1 -2
- data/lib/twitter_cldr/resources/custom_locales_resources_importer.rb +4 -4
- data/lib/twitter_cldr/resources/download.rb +13 -6
- data/lib/twitter_cldr/resources/language_codes_importer.rb +7 -7
- data/lib/twitter_cldr/resources/loader.rb +4 -1
- data/lib/twitter_cldr/resources/locales_resources_importer.rb +30 -12
- data/lib/twitter_cldr/resources/postal_codes_importer.rb +32 -22
- data/lib/twitter_cldr/resources/properties/age_property_importer.rb +27 -0
- data/lib/twitter_cldr/resources/properties/arabic_shaping_property_importer.rb +42 -0
- data/lib/twitter_cldr/resources/properties/bidi_brackets_property_importer.rb +41 -0
- data/lib/twitter_cldr/resources/properties/blocks_property_importer.rb +27 -0
- data/lib/twitter_cldr/resources/properties/derived_core_properties_importer.rb +36 -0
- data/lib/twitter_cldr/resources/properties/east_asian_width_property_importer.rb +27 -0
- data/lib/twitter_cldr/resources/properties/grapheme_break_property_importer.rb +27 -0
- data/lib/twitter_cldr/resources/properties/hangul_syllable_type_property_importer.rb +27 -0
- data/lib/twitter_cldr/resources/properties/indic_positional_category_property_importer.rb +27 -0
- data/lib/twitter_cldr/resources/properties/indic_syllabic_category_property_importer.rb +27 -0
- data/lib/twitter_cldr/resources/properties/jamo_property_importer.rb +27 -0
- data/lib/twitter_cldr/resources/properties/line_break_property_importer.rb +27 -0
- data/lib/twitter_cldr/resources/properties/prop_list_importer.rb +36 -0
- data/lib/twitter_cldr/resources/properties/properties_importer.rb +59 -0
- data/lib/twitter_cldr/resources/properties/property_importer.rb +83 -0
- data/lib/twitter_cldr/resources/properties/script_extensions_property_importer.rb +40 -0
- data/lib/twitter_cldr/resources/properties/script_property_importer.rb +27 -0
- data/lib/twitter_cldr/resources/properties/sentence_break_property_importer.rb +27 -0
- data/lib/twitter_cldr/resources/properties/unicode_data_properties_importer.rb +60 -0
- data/lib/twitter_cldr/resources/properties/word_break_property_importer.rb +27 -0
- data/lib/twitter_cldr/resources/properties.rb +36 -0
- data/lib/twitter_cldr/resources/readme_renderer.rb +2 -2
- data/lib/twitter_cldr/resources/segment_tests_importer.rb +66 -0
- data/lib/twitter_cldr/resources/tailoring_importer.rb +7 -7
- data/lib/twitter_cldr/resources/uli/segment_exceptions_importer.rb +1 -1
- data/lib/twitter_cldr/resources/unicode_data_importer.rb +19 -60
- data/lib/twitter_cldr/resources/unicode_property_aliases_importer.rb +97 -0
- data/lib/twitter_cldr/resources.rb +21 -22
- data/lib/twitter_cldr/segmentation/break_iterator.rb +54 -0
- data/lib/twitter_cldr/segmentation/cursor.rb +34 -0
- data/lib/twitter_cldr/segmentation/parser.rb +71 -0
- data/lib/twitter_cldr/segmentation/rule.rb +79 -0
- data/lib/twitter_cldr/segmentation/rule_set.rb +116 -0
- data/lib/twitter_cldr/segmentation/rule_set_builder.rb +142 -0
- data/lib/twitter_cldr/segmentation.rb +17 -0
- data/lib/twitter_cldr/shared/bidi.rb +4 -4
- data/lib/twitter_cldr/shared/calendar.rb +11 -11
- data/lib/twitter_cldr/shared/caser.rb +84 -0
- data/lib/twitter_cldr/shared/code_point.rb +101 -139
- data/lib/twitter_cldr/shared/currencies.rb +5 -5
- data/lib/twitter_cldr/shared/language_codes.rb +2 -2
- data/lib/twitter_cldr/shared/languages.rb +1 -1
- data/lib/twitter_cldr/shared/likely_subtags.rb +104 -0
- data/lib/twitter_cldr/shared/locale.rb +252 -0
- data/lib/twitter_cldr/shared/postal_codes.rb +21 -9
- data/lib/twitter_cldr/shared/properties/arabic_shaping.rb +40 -0
- data/lib/twitter_cldr/shared/properties/bidi_brackets.rb +28 -0
- data/lib/twitter_cldr/shared/properties.rb +13 -0
- data/lib/twitter_cldr/shared/properties_database.rb +180 -0
- data/lib/twitter_cldr/shared/property_name_aliases.rb +48 -0
- data/lib/twitter_cldr/shared/property_normalizer.rb +108 -0
- data/lib/twitter_cldr/shared/property_set.rb +113 -0
- data/lib/twitter_cldr/shared/property_value_aliases.rb +99 -0
- data/lib/twitter_cldr/shared/unicode_regex.rb +29 -3
- data/lib/twitter_cldr/shared.rb +9 -1
- data/lib/twitter_cldr/tokenizers/numbers/number_tokenizer.rb +1 -1
- data/lib/twitter_cldr/tokenizers/token.rb +1 -1
- data/lib/twitter_cldr/tokenizers/tokenizer.rb +15 -13
- data/lib/twitter_cldr/tokenizers/unicode_regex/unicode_regex_tokenizer.rb +4 -4
- data/lib/twitter_cldr/utils/file_system_trie.rb +145 -0
- data/lib/twitter_cldr/utils/range_set.rb +131 -31
- data/lib/twitter_cldr/utils/regexp_sampler.rb +0 -1
- data/lib/twitter_cldr/utils/script_detector.rb +75 -0
- data/lib/twitter_cldr/utils/yaml.rb +3 -3
- data/lib/twitter_cldr/utils.rb +8 -5
- data/lib/twitter_cldr/version.rb +1 -1
- data/lib/twitter_cldr/versions.rb +30 -0
- data/lib/twitter_cldr.rb +8 -10
- data/resources/locales/af/calendars.yml +8 -8
- data/resources/locales/af/lists.yml +8 -4
- data/resources/locales/af/numbers.yml +54 -30
- data/resources/locales/ar/lists.yml +8 -4
- data/resources/locales/ar/numbers.yml +48 -24
- data/resources/locales/be/calendars.yml +4 -4
- data/resources/locales/be/lists.yml +2 -1
- data/resources/locales/be/numbers.yml +24 -12
- data/resources/locales/bg/calendars.yml +8 -8
- data/resources/locales/bg/lists.yml +8 -4
- data/resources/locales/bg/numbers.yml +48 -24
- data/resources/locales/bn/lists.yml +8 -4
- data/resources/locales/bn/numbers.yml +48 -24
- data/resources/locales/ca/calendars.yml +8 -8
- data/resources/locales/ca/lists.yml +8 -4
- data/resources/locales/ca/numbers.yml +48 -24
- data/resources/locales/cs/calendars.yml +32 -32
- data/resources/locales/cs/lists.yml +8 -4
- data/resources/locales/cs/numbers.yml +48 -24
- data/resources/locales/cy/calendars.yml +8 -8
- data/resources/locales/cy/lists.yml +6 -3
- data/resources/locales/cy/numbers.yml +48 -24
- data/resources/locales/da/calendars.yml +10 -10
- data/resources/locales/da/lists.yml +8 -4
- data/resources/locales/da/numbers.yml +48 -24
- data/resources/locales/de/calendars.yml +8 -8
- data/resources/locales/de/lists.yml +8 -4
- data/resources/locales/de/numbers.yml +48 -24
- data/resources/locales/de-CH/calendars.yml +8 -8
- data/resources/locales/de-CH/lists.yml +8 -4
- data/resources/locales/de-CH/numbers.yml +48 -24
- data/resources/locales/el/calendars.yml +8 -8
- data/resources/locales/el/lists.yml +8 -4
- data/resources/locales/el/numbers.yml +48 -24
- data/resources/locales/en/calendars.yml +4 -4
- data/resources/locales/en/lists.yml +8 -4
- data/resources/locales/en/numbers.yml +48 -24
- data/resources/locales/en-150/calendars.yml +4 -4
- data/resources/locales/en-150/lists.yml +8 -4
- data/resources/locales/en-150/numbers.yml +48 -24
- data/resources/locales/en-AU/calendars.yml +4 -4
- data/resources/locales/en-AU/lists.yml +8 -4
- data/resources/locales/en-AU/numbers.yml +48 -24
- data/resources/locales/en-CA/calendars.yml +4 -4
- data/resources/locales/en-CA/lists.yml +8 -4
- data/resources/locales/en-CA/numbers.yml +48 -24
- data/resources/locales/en-GB/calendars.yml +4 -4
- data/resources/locales/en-GB/lists.yml +8 -4
- data/resources/locales/en-GB/numbers.yml +48 -24
- data/resources/locales/en-IE/calendars.yml +4 -4
- data/resources/locales/en-IE/lists.yml +8 -4
- data/resources/locales/en-IE/numbers.yml +48 -24
- data/resources/locales/en-SG/calendars.yml +4 -4
- data/resources/locales/en-SG/lists.yml +8 -4
- data/resources/locales/en-SG/numbers.yml +48 -24
- data/resources/locales/en-ZA/calendars.yml +4 -4
- data/resources/locales/en-ZA/lists.yml +8 -4
- data/resources/locales/en-ZA/numbers.yml +48 -24
- data/resources/locales/es/lists.yml +8 -4
- data/resources/locales/es/numbers.yml +48 -24
- data/resources/locales/es-419/calendars.yml +8 -8
- data/resources/locales/es-419/lists.yml +8 -4
- data/resources/locales/es-419/numbers.yml +50 -26
- data/resources/locales/es-CO/lists.yml +8 -4
- data/resources/locales/es-CO/numbers.yml +48 -24
- data/resources/locales/es-MX/lists.yml +8 -4
- data/resources/locales/es-MX/numbers.yml +48 -24
- data/resources/locales/es-US/lists.yml +8 -4
- data/resources/locales/es-US/numbers.yml +48 -24
- data/resources/locales/eu/calendars.yml +8 -8
- data/resources/locales/eu/lists.yml +8 -4
- data/resources/locales/eu/numbers.yml +48 -24
- data/resources/locales/fa/lists.yml +8 -4
- data/resources/locales/fa/numbers.yml +48 -24
- data/resources/locales/fi/calendars.yml +8 -8
- data/resources/locales/fi/lists.yml +8 -4
- data/resources/locales/fi/numbers.yml +48 -24
- data/resources/locales/fil/calendars.yml +8 -8
- data/resources/locales/fil/lists.yml +8 -4
- data/resources/locales/fil/numbers.yml +48 -24
- data/resources/locales/fr/calendars.yml +8 -8
- data/resources/locales/fr/lists.yml +8 -4
- data/resources/locales/fr/numbers.yml +48 -24
- data/resources/locales/fr-BE/calendars.yml +8 -8
- data/resources/locales/fr-BE/lists.yml +8 -4
- data/resources/locales/fr-BE/numbers.yml +48 -24
- data/resources/locales/fr-CA/calendars.yml +8 -8
- data/resources/locales/fr-CA/lists.yml +8 -4
- data/resources/locales/fr-CA/numbers.yml +48 -24
- data/resources/locales/fr-CH/calendars.yml +8 -8
- data/resources/locales/fr-CH/lists.yml +8 -4
- data/resources/locales/fr-CH/numbers.yml +48 -24
- data/resources/locales/ga/calendars.yml +8 -8
- data/resources/locales/ga/lists.yml +8 -4
- data/resources/locales/ga/numbers.yml +48 -24
- data/resources/locales/gl/calendars.yml +8 -8
- data/resources/locales/gl/lists.yml +8 -4
- data/resources/locales/gl/numbers.yml +48 -24
- data/resources/locales/he/calendars.yml +28 -28
- data/resources/locales/he/lists.yml +8 -4
- data/resources/locales/he/numbers.yml +48 -24
- data/resources/locales/hi/calendars.yml +8 -8
- data/resources/locales/hi/lists.yml +8 -4
- data/resources/locales/hi/numbers.yml +48 -24
- data/resources/locales/hr/lists.yml +8 -4
- data/resources/locales/hr/numbers.yml +48 -24
- data/resources/locales/hu/lists.yml +8 -4
- data/resources/locales/hu/numbers.yml +48 -24
- data/resources/locales/id/calendars.yml +8 -8
- data/resources/locales/id/lists.yml +8 -4
- data/resources/locales/id/numbers.yml +49 -25
- data/resources/locales/is/calendars.yml +8 -8
- data/resources/locales/is/lists.yml +8 -4
- data/resources/locales/is/numbers.yml +48 -24
- data/resources/locales/it/calendars.yml +8 -8
- data/resources/locales/it/lists.yml +8 -4
- data/resources/locales/it/numbers.yml +54 -30
- data/resources/locales/it-CH/calendars.yml +8 -8
- data/resources/locales/it-CH/lists.yml +8 -4
- data/resources/locales/it-CH/numbers.yml +54 -30
- data/resources/locales/ja/calendars.yml +32 -32
- data/resources/locales/ja/lists.yml +8 -4
- data/resources/locales/ja/numbers.yml +48 -24
- data/resources/locales/ko/calendars.yml +8 -8
- data/resources/locales/ko/lists.yml +8 -4
- data/resources/locales/ko/numbers.yml +48 -24
- data/resources/locales/lv/lists.yml +8 -4
- data/resources/locales/lv/numbers.yml +48 -24
- data/resources/locales/ms/calendars.yml +8 -8
- data/resources/locales/ms/lists.yml +8 -4
- data/resources/locales/ms/numbers.yml +48 -24
- data/resources/locales/nb/calendars.yml +9 -9
- data/resources/locales/nb/lists.yml +8 -4
- data/resources/locales/nb/numbers.yml +48 -24
- data/resources/locales/nl/calendars.yml +8 -8
- data/resources/locales/nl/lists.yml +8 -4
- data/resources/locales/nl/numbers.yml +48 -24
- data/resources/locales/pl/calendars.yml +8 -8
- data/resources/locales/pl/lists.yml +8 -4
- data/resources/locales/pl/numbers.yml +48 -24
- data/resources/locales/pt/calendars.yml +8 -8
- data/resources/locales/pt/lists.yml +8 -4
- data/resources/locales/pt/numbers.yml +48 -24
- data/resources/locales/ro/calendars.yml +8 -8
- data/resources/locales/ro/lists.yml +8 -4
- data/resources/locales/ro/numbers.yml +48 -24
- data/resources/locales/ru/calendars.yml +8 -8
- data/resources/locales/ru/lists.yml +8 -4
- data/resources/locales/ru/numbers.yml +48 -24
- data/resources/locales/sk/calendars.yml +8 -8
- data/resources/locales/sk/lists.yml +8 -4
- data/resources/locales/sk/numbers.yml +48 -24
- data/resources/locales/sq/calendars.yml +8 -8
- data/resources/locales/sq/lists.yml +8 -4
- data/resources/locales/sq/numbers.yml +48 -24
- data/resources/locales/sr/lists.yml +8 -4
- data/resources/locales/sr/numbers.yml +48 -24
- data/resources/locales/sv/calendars.yml +10 -10
- data/resources/locales/sv/lists.yml +8 -4
- data/resources/locales/sv/numbers.yml +48 -24
- data/resources/locales/ta/calendars.yml +8 -8
- data/resources/locales/ta/lists.yml +8 -4
- data/resources/locales/ta/numbers.yml +48 -24
- data/resources/locales/th/calendars.yml +8 -8
- data/resources/locales/th/lists.yml +8 -4
- data/resources/locales/th/numbers.yml +48 -24
- data/resources/locales/tr/lists.yml +8 -4
- data/resources/locales/tr/numbers.yml +50 -26
- data/resources/locales/uk/calendars.yml +8 -8
- data/resources/locales/uk/lists.yml +8 -4
- data/resources/locales/uk/numbers.yml +48 -24
- data/resources/locales/ur/calendars.yml +8 -8
- data/resources/locales/ur/lists.yml +8 -4
- data/resources/locales/ur/numbers.yml +48 -24
- data/resources/locales/vi/calendars.yml +32 -32
- data/resources/locales/vi/lists.yml +8 -4
- data/resources/locales/vi/numbers.yml +48 -24
- data/resources/locales/zh/calendars.yml +32 -32
- data/resources/locales/zh/lists.yml +8 -4
- data/resources/locales/zh/numbers.yml +48 -24
- data/resources/locales/zh-Hant/calendars.yml +32 -32
- data/resources/locales/zh-Hant/lists.yml +8 -4
- data/resources/locales/zh-Hant/numbers.yml +48 -24
- data/resources/shared/aliases.yml +1351 -0
- data/resources/shared/likely_subtags.yml +1149 -0
- data/resources/shared/postal_codes.yml +486 -289
- data/resources/shared/segments/segments_root.yml +57 -57
- data/resources/shared/segments/tests/sentence_break_test.yml +527 -0
- data/resources/shared/segments/tests/word_break_test.yml +1379 -0
- data/resources/shared/territories_containment.yml +25 -17
- data/resources/shared/variables.yml +1194 -0
- data/resources/unicode_data/blocks/ahom.yml +913 -0
- data/resources/unicode_data/blocks/anatolian_hieroglyphs.yml +9329 -0
- data/resources/unicode_data/blocks/arabic.yml +16 -0
- data/resources/unicode_data/blocks/arabic_presentation_forms_a.yml +2 -2
- data/resources/unicode_data/blocks/bassa_vah.yml +577 -0
- data/resources/unicode_data/blocks/buginese.yml +2 -2
- data/resources/unicode_data/blocks/caucasian_albanian.yml +849 -0
- data/resources/unicode_data/blocks/cherokee_supplement.yml +1281 -0
- data/resources/unicode_data/blocks/cjk_unified_ideographs_extension_e.yml +33 -0
- data/resources/unicode_data/blocks/combining_diacritical_marks_extended.yml +241 -0
- data/resources/unicode_data/blocks/coptic_epact_numbers.yml +449 -0
- data/resources/unicode_data/blocks/cuneiform_numbers_and_punctuation.yml +2 -2
- data/resources/unicode_data/blocks/duployan.yml +2289 -0
- data/resources/unicode_data/blocks/early_dynastic_cuneiform.yml +3137 -0
- data/resources/unicode_data/blocks/elbasan.yml +641 -0
- data/resources/unicode_data/blocks/general_punctuation.yml +64 -0
- data/resources/unicode_data/blocks/geometric_shapes_extended.yml +1361 -0
- data/resources/unicode_data/blocks/grantha.yml +1361 -0
- data/resources/unicode_data/blocks/gujarati.yml +0 -16
- data/resources/unicode_data/blocks/hatran.yml +417 -0
- data/resources/unicode_data/blocks/kannada.yml +0 -16
- data/resources/unicode_data/blocks/khojki.yml +977 -0
- data/resources/unicode_data/blocks/khudawadi.yml +1105 -0
- data/resources/unicode_data/blocks/latin_extended_e.yml +865 -0
- data/resources/unicode_data/blocks/linear_a.yml +5457 -0
- data/resources/unicode_data/blocks/mahajani.yml +625 -0
- data/resources/unicode_data/blocks/manichaean.yml +817 -0
- data/resources/unicode_data/blocks/mende_kikakui.yml +3409 -0
- data/resources/unicode_data/blocks/miscellaneous_technical.yml +4 -4
- data/resources/unicode_data/blocks/modi.yml +1265 -0
- data/resources/unicode_data/blocks/mongolian.yml +2 -2
- data/resources/unicode_data/blocks/mro.yml +689 -0
- data/resources/unicode_data/blocks/multani.yml +609 -0
- data/resources/unicode_data/blocks/myanmar_extended_b.yml +497 -0
- data/resources/unicode_data/blocks/nabataean.yml +641 -0
- data/resources/unicode_data/blocks/old_hungarian.yml +1729 -0
- data/resources/unicode_data/blocks/old_north_arabian.yml +513 -0
- data/resources/unicode_data/blocks/old_permic.yml +689 -0
- data/resources/unicode_data/blocks/ornamental_dingbats.yml +769 -0
- data/resources/unicode_data/blocks/pahawh_hmong.yml +2033 -0
- data/resources/unicode_data/blocks/palmyrene.yml +513 -0
- data/resources/unicode_data/blocks/pau_cin_hau.yml +913 -0
- data/resources/unicode_data/blocks/psalter_pahlavi.yml +465 -0
- data/resources/unicode_data/blocks/shorthand_format_controls.yml +65 -0
- data/resources/unicode_data/blocks/siddham.yml +1473 -0
- data/resources/unicode_data/blocks/sinhala_archaic_numbers.yml +321 -0
- data/resources/unicode_data/blocks/supplemental_arrows_c.yml +2369 -0
- data/resources/unicode_data/blocks/supplemental_symbols_and_pictographs.yml +241 -0
- data/resources/unicode_data/blocks/sutton_signwriting.yml +10753 -0
- data/resources/unicode_data/blocks/tirhuta.yml +1313 -0
- data/resources/unicode_data/blocks/warang_citi.yml +1345 -0
- data/resources/unicode_data/properties/ASCII_Hex_Digit/value.dump +5 -0
- data/resources/unicode_data/properties/Age/1.1/value.dump +0 -0
- data/resources/unicode_data/properties/Age/2.0/value.dump +0 -0
- data/resources/unicode_data/properties/Age/2.1/value.dump +4 -0
- data/resources/unicode_data/properties/Age/3.0/value.dump +0 -0
- data/resources/unicode_data/properties/Age/3.1/value.dump +0 -0
- data/resources/unicode_data/properties/Age/3.2/value.dump +0 -0
- data/resources/unicode_data/properties/Age/4.0/value.dump +0 -0
- data/resources/unicode_data/properties/Age/4.1/value.dump +0 -0
- data/resources/unicode_data/properties/Age/5.0/value.dump +0 -0
- data/resources/unicode_data/properties/Age/5.1/value.dump +0 -0
- data/resources/unicode_data/properties/Age/5.2/value.dump +0 -0
- data/resources/unicode_data/properties/Age/6.0/value.dump +0 -0
- data/resources/unicode_data/properties/Age/6.1/value.dump +0 -0
- data/resources/unicode_data/properties/Age/6.2/value.dump +3 -0
- data/resources/unicode_data/properties/Age/6.3/value.dump +4 -0
- data/resources/unicode_data/properties/Alphabetic/value.dump +0 -0
- data/resources/unicode_data/properties/Bidi_Class/AL/value.dump +0 -0
- data/resources/unicode_data/properties/Bidi_Class/AN/value.dump +0 -0
- data/resources/unicode_data/properties/Bidi_Class/B/value.dump +8 -0
- data/resources/unicode_data/properties/Bidi_Class/BN/value.dump +0 -0
- data/resources/unicode_data/properties/Bidi_Class/CS/value.dump +15 -0
- data/resources/unicode_data/properties/Bidi_Class/EN/value.dump +0 -0
- data/resources/unicode_data/properties/Bidi_Class/ES/value.dump +11 -0
- data/resources/unicode_data/properties/Bidi_Class/ET/value.dump +27 -0
- data/resources/unicode_data/properties/Bidi_Class/FSI/value.dump +3 -0
- data/resources/unicode_data/properties/Bidi_Class/L/value.dump +0 -0
- data/resources/unicode_data/properties/Bidi_Class/LRE/value.dump +3 -0
- data/resources/unicode_data/properties/Bidi_Class/LRI/value.dump +3 -0
- data/resources/unicode_data/properties/Bidi_Class/LRO/value.dump +3 -0
- data/resources/unicode_data/properties/Bidi_Class/NSM/value.dump +0 -0
- data/resources/unicode_data/properties/Bidi_Class/ON/value.dump +0 -0
- data/resources/unicode_data/properties/Bidi_Class/PDF/value.dump +3 -0
- data/resources/unicode_data/properties/Bidi_Class/PDI/value.dump +3 -0
- data/resources/unicode_data/properties/Bidi_Class/R/value.dump +0 -0
- data/resources/unicode_data/properties/Bidi_Class/RLE/value.dump +3 -0
- data/resources/unicode_data/properties/Bidi_Class/RLI/value.dump +3 -0
- data/resources/unicode_data/properties/Bidi_Class/RLO/value.dump +3 -0
- data/resources/unicode_data/properties/Bidi_Class/S/value.dump +5 -0
- data/resources/unicode_data/properties/Bidi_Class/WS/value.dump +0 -0
- data/resources/unicode_data/properties/Bidi_Control/value.dump +4 -0
- data/resources/unicode_data/properties/Bidi_Mirrored/N/value.dump +0 -0
- data/resources/unicode_data/properties/Bidi_Mirrored/Y/value.dump +115 -0
- data/resources/unicode_data/properties/Block/Aegean Numbers/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Alchemical Symbols/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Alphabetic Presentation Forms/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Ancient Greek Musical Notation/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Ancient Greek Numbers/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Ancient Symbols/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Arabic/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Arabic Extended-A/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Arabic Mathematical Alphabetic Symbols/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Arabic Presentation Forms-A/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Arabic Presentation Forms-B/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Arabic Supplement/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Armenian/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Arrows/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Avestan/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Balinese/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Bamum/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Bamum Supplement/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Basic Latin/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Batak/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Bengali/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Block Elements/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Bopomofo/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Bopomofo Extended/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Box Drawing/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Brahmi/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Braille Patterns/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Buginese/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Buhid/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Byzantine Musical Symbols/value.dump +0 -0
- data/resources/unicode_data/properties/Block/CJK Compatibility/value.dump +0 -0
- data/resources/unicode_data/properties/Block/CJK Compatibility Forms/value.dump +3 -0
- data/resources/unicode_data/properties/Block/CJK Compatibility Ideographs/value.dump +0 -0
- data/resources/unicode_data/properties/Block/CJK Compatibility Ideographs Supplement/value.dump +0 -0
- data/resources/unicode_data/properties/Block/CJK Radicals Supplement/value.dump +3 -0
- data/resources/unicode_data/properties/Block/CJK Strokes/value.dump +3 -0
- Punctuation/value.dump +0 -0
- data/resources/unicode_data/properties/Block/CJK Unified Ideographs/value.dump +0 -0
- data/resources/unicode_data/properties/Block/CJK Unified Ideographs Extension A/value.dump +0 -0
- data/resources/unicode_data/properties/Block/CJK Unified Ideographs Extension B/value.dump +0 -0
- data/resources/unicode_data/properties/Block/CJK Unified Ideographs Extension C/value.dump +0 -0
- data/resources/unicode_data/properties/Block/CJK Unified Ideographs Extension D/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Carian/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Chakma/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Cham/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Cherokee/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Combining Diacritical Marks/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Combining Diacritical Marks Supplement/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Combining Diacritical Marks for Symbols/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Combining Half Marks/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Common Indic Number Forms/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Control Pictures/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Coptic/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Counting Rod Numerals/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Cuneiform/value.dump +0 -0
- Punctuation/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Currency Symbols/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Cypriot Syllabary/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Cyrillic/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Cyrillic Extended-A/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Cyrillic Extended-B/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Cyrillic Supplement/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Deseret/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Devanagari/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Devanagari Extended/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Dingbats/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Domino Tiles/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Egyptian Hieroglyphs/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Emoticons/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Enclosed Alphanumeric Supplement/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Enclosed Alphanumerics/value.dump +3 -0
- Months/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Enclosed Ideographic Supplement/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Ethiopic/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Ethiopic Extended/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Ethiopic Extended-A/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Ethiopic Supplement/value.dump +3 -0
- data/resources/unicode_data/properties/Block/General Punctuation/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Geometric Shapes/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Georgian/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Georgian Supplement/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Glagolitic/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Gothic/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Greek Extended/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Greek and Coptic/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Gujarati/value.dump +4 -0
- data/resources/unicode_data/properties/Block/Gurmukhi/value.dump +0 -0
- Fullwidth Forms/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Hangul Compatibility Jamo/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Hangul Jamo/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Hangul Jamo Extended-A/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Hangul Jamo Extended-B/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Hangul Syllables/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Hanunoo/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Hebrew/value.dump +3 -0
- data/resources/unicode_data/properties/Block/High Private Use Surrogates/value.dump +3 -0
- data/resources/unicode_data/properties/Block/High Surrogates/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Hiragana/value.dump +3 -0
- data/resources/unicode_data/properties/Block/IPA Extensions/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Ideographic Description Characters/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Imperial Aramaic/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Inscriptional Pahlavi/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Inscriptional Parthian/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Javanese/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Kaithi/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Kana Supplement/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Kanbun/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Kangxi Radicals/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Kannada/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Katakana/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Katakana Phonetic Extensions/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Kayah Li/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Kharoshthi/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Khmer/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Khmer Symbols/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Lao/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Latin Extended Additional/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Latin Extended-A/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Latin Extended-B/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Latin Extended-C/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Latin Extended-D/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Latin-1 Supplement/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Lepcha/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Letterlike Symbols/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Limbu/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Linear B Ideograms/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Linear B Syllabary/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Lisu/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Low Surrogates/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Lycian/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Lydian/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Mahjong Tiles/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Malayalam/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Mandaic/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Mathematical Alphanumeric Symbols/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Mathematical Operators/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Meetei Mayek/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Meetei Mayek Extensions/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Meroitic Cursive/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Meroitic Hieroglyphs/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Miao/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Miscellaneous Mathematical Symbols-A/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Miscellaneous Mathematical Symbols-B/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Miscellaneous Symbols/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Miscellaneous Symbols And Pictographs/value.dump +0 -0
- Arrows/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Miscellaneous Technical/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Modifier Tone Letters/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Mongolian/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Musical Symbols/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Myanmar/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Myanmar Extended-A/value.dump +3 -0
- data/resources/unicode_data/properties/Block/NKo/value.dump +3 -0
- data/resources/unicode_data/properties/Block/New Tai Lue/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Number Forms/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Ogham/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Ol Chiki/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Old Italic/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Old Persian/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Old South Arabian/value.dump +5 -0
- data/resources/unicode_data/properties/Block/Old Turkic/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Optical Character Recognition/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Oriya/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Osmanya/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Phags-pa/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Phaistos Disc/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Phoenician/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Phonetic Extensions/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Phonetic Extensions Supplement/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Playing Cards/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Private Use Area/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Rejang/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Rumi Numeral Symbols/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Runic/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Samaritan/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Saurashtra/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Sharada/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Shavian/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Sinhala/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Small Form Variants/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Sora Sompeng/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Spacing Modifier Letters/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Specials/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Sundanese/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Sundanese Supplement/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Superscripts and Subscripts/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Supplemental Arrows-A/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Supplemental Arrows-B/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Supplemental Mathematical Operators/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Supplemental Punctuation/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Supplementary Private Use Area-A/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Supplementary Private Use Area-B/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Syloti Nagri/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Syriac/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Tagalog/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Tagbanwa/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Tags/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Tai Le/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Tai Tham/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Tai Viet/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Tai Xuan Jing Symbols/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Takri/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Tamil/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Telugu/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Thaana/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Thai/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Tibetan/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Tifinagh/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Transport And Map Symbols/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Ugaritic/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Unified Canadian Aboriginal Syllabics/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Unified Canadian Aboriginal Syllabics Extended/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Vai/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Variation Selectors/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Variation Selectors Supplement/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Vedic Extensions/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Vertical Forms/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Yi Radicals/value.dump +3 -0
- data/resources/unicode_data/properties/Block/Yi Syllables/value.dump +0 -0
- data/resources/unicode_data/properties/Block/Yijing Hexagram Symbols/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/0/value.dump +0 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/1/value.dump +13 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/10/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/103/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/107/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/11/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/118/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/12/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/122/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/129/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/13/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/130/value.dump +5 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/132/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/14/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/15/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/16/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/17/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/18/value.dump +4 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/19/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/20/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/202/value.dump +5 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/21/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/214/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/216/value.dump +6 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/218/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/22/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/220/value.dump +70 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/222/value.dump +6 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/224/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/226/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/228/value.dump +5 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/23/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/230/value.dump +0 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/232/value.dump +6 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/233/value.dump +6 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/234/value.dump +5 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/24/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/240/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/25/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/26/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/27/value.dump +4 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/28/value.dump +4 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/29/value.dump +4 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/30/value.dump +4 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/31/value.dump +4 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/32/value.dump +4 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/33/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/34/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/35/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/36/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/7/value.dump +19 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/8/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/84/value.dump +3 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/9/value.dump +41 -0
- data/resources/unicode_data/properties/Canonical_Combining_Class/91/value.dump +3 -0
- data/resources/unicode_data/properties/Case_Ignorable/value.dump +0 -0
- data/resources/unicode_data/properties/Cased/value.dump +0 -0
- data/resources/unicode_data/properties/Changes_When_Casefolded/value.dump +0 -0
- data/resources/unicode_data/properties/Changes_When_Casemapped/value.dump +0 -0
- data/resources/unicode_data/properties/Changes_When_Lowercased/value.dump +0 -0
- data/resources/unicode_data/properties/Changes_When_Titlecased/value.dump +589 -0
- data/resources/unicode_data/properties/Changes_When_Uppercased/value.dump +588 -0
- data/resources/unicode_data/properties/Dash/value.dump +0 -0
- data/resources/unicode_data/properties/Default_Ignorable_Code_Point/value.dump +0 -0
- data/resources/unicode_data/properties/Deprecated/value.dump +0 -0
- data/resources/unicode_data/properties/Diacritic/value.dump +135 -0
- data/resources/unicode_data/properties/East_Asian_Width/A/value.dump +0 -0
- data/resources/unicode_data/properties/East_Asian_Width/F/value.dump +0 -0
- data/resources/unicode_data/properties/East_Asian_Width/H/value.dump +9 -0
- data/resources/unicode_data/properties/East_Asian_Width/N/value.dump +0 -0
- data/resources/unicode_data/properties/East_Asian_Width/Na/value.dump +9 -0
- data/resources/unicode_data/properties/East_Asian_Width/W/value.dump +0 -0
- data/resources/unicode_data/properties/Extender/value.dump +26 -0
- data/resources/unicode_data/properties/General_Category/C/c/value.dump +0 -0
- data/resources/unicode_data/properties/General_Category/C/f/value.dump +0 -0
- data/resources/unicode_data/properties/General_Category/C/o/value.dump +0 -0
- data/resources/unicode_data/properties/General_Category/C/s/value.dump +0 -0
- data/resources/unicode_data/properties/General_Category/L/l/value.dump +0 -0
- data/resources/unicode_data/properties/General_Category/L/m/value.dump +54 -0
- data/resources/unicode_data/properties/General_Category/L/o/value.dump +0 -0
- data/resources/unicode_data/properties/General_Category/L/t/value.dump +12 -0
- data/resources/unicode_data/properties/General_Category/L/u/value.dump +0 -0
- data/resources/unicode_data/properties/General_Category/M/c/value.dump +0 -0
- data/resources/unicode_data/properties/General_Category/M/e/value.dump +6 -0
- data/resources/unicode_data/properties/General_Category/M/n/value.dump +0 -0
- data/resources/unicode_data/properties/General_Category/N/d/value.dump +0 -0
- data/resources/unicode_data/properties/General_Category/N/l/value.dump +0 -0
- data/resources/unicode_data/properties/General_Category/N/o/value.dump +0 -0
- data/resources/unicode_data/properties/General_Category/P/c/value.dump +8 -0
- data/resources/unicode_data/properties/General_Category/P/d/value.dump +0 -0
- data/resources/unicode_data/properties/General_Category/P/e/value.dump +72 -0
- data/resources/unicode_data/properties/General_Category/P/f/value.dump +14 -0
- data/resources/unicode_data/properties/General_Category/P/i/value.dump +13 -0
- data/resources/unicode_data/properties/General_Category/P/o/value.dump +0 -0
- data/resources/unicode_data/properties/General_Category/P/s/value.dump +76 -0
- data/resources/unicode_data/properties/General_Category/S/c/value.dump +21 -0
- data/resources/unicode_data/properties/General_Category/S/k/value.dump +0 -0
- data/resources/unicode_data/properties/General_Category/S/m/value.dump +0 -0
- data/resources/unicode_data/properties/General_Category/S/o/value.dump +0 -0
- data/resources/unicode_data/properties/General_Category/Z/l/value.dump +3 -0
- data/resources/unicode_data/properties/General_Category/Z/p/value.dump +3 -0
- data/resources/unicode_data/properties/General_Category/Z/s/value.dump +0 -0
- data/resources/unicode_data/properties/Grapheme_Base/value.dump +0 -0
- data/resources/unicode_data/properties/Grapheme_Cluster_Break/CR/value.dump +3 -0
- data/resources/unicode_data/properties/Grapheme_Cluster_Break/Control/value.dump +0 -0
- data/resources/unicode_data/properties/Grapheme_Cluster_Break/Extend/value.dump +0 -0
- data/resources/unicode_data/properties/Grapheme_Cluster_Break/L/value.dump +0 -0
- data/resources/unicode_data/properties/Grapheme_Cluster_Break/LF/value.dump +3 -0
- data/resources/unicode_data/properties/Grapheme_Cluster_Break/LV/value.dump +0 -0
- data/resources/unicode_data/properties/Grapheme_Cluster_Break/LVT/value.dump +401 -0
- data/resources/unicode_data/properties/Grapheme_Cluster_Break/Regional_Indicator/value.dump +3 -0
- data/resources/unicode_data/properties/Grapheme_Cluster_Break/SpacingMark/value.dump +0 -0
- data/resources/unicode_data/properties/Grapheme_Cluster_Break/T/value.dump +4 -0
- data/resources/unicode_data/properties/Grapheme_Cluster_Break/V/value.dump +4 -0
- data/resources/unicode_data/properties/Grapheme_Extend/value.dump +0 -0
- data/resources/unicode_data/properties/Grapheme_Link/value.dump +41 -0
- data/resources/unicode_data/properties/Hangul_Syllable_Type/L/value.dump +0 -0
- data/resources/unicode_data/properties/Hangul_Syllable_Type/LV/value.dump +0 -0
- data/resources/unicode_data/properties/Hangul_Syllable_Type/LVT/value.dump +401 -0
- data/resources/unicode_data/properties/Hangul_Syllable_Type/T/value.dump +4 -0
- data/resources/unicode_data/properties/Hangul_Syllable_Type/V/value.dump +4 -0
- data/resources/unicode_data/properties/Hex_Digit/value.dump +8 -0
- data/resources/unicode_data/properties/Hyphen/value.dump +12 -0
- data/resources/unicode_data/properties/IDS_Binary_Operator/value.dump +4 -0
- data/resources/unicode_data/properties/IDS_Trinary_Operator/value.dump +3 -0
- data/resources/unicode_data/properties/ID_Continue/value.dump +0 -0
- data/resources/unicode_data/properties/ID_Start/value.dump +0 -0
- data/resources/unicode_data/properties/Ideographic/value.dump +0 -0
- data/resources/unicode_data/properties/Indic_Positional_Category/Bottom/value.dump +78 -0
- data/resources/unicode_data/properties/Indic_Positional_Category/Bottom_And_Right/value.dump +4 -0
- data/resources/unicode_data/properties/Indic_Positional_Category/Invisible/value.dump +10 -0
- data/resources/unicode_data/properties/Indic_Positional_Category/Left/value.dump +36 -0
- data/resources/unicode_data/properties/Indic_Positional_Category/Left_And_Right/value.dump +10 -0
- data/resources/unicode_data/properties/Indic_Positional_Category/Overstruck/value.dump +8 -0
- data/resources/unicode_data/properties/Indic_Positional_Category/Right/value.dump +86 -0
- data/resources/unicode_data/properties/Indic_Positional_Category/Top/value.dump +91 -0
- data/resources/unicode_data/properties/Indic_Positional_Category/Top_And_Bottom/value.dump +8 -0
- data/resources/unicode_data/properties/Indic_Positional_Category/Top_And_Bottom_And_Right/value.dump +3 -0
- data/resources/unicode_data/properties/Indic_Positional_Category/Top_And_Left/value.dump +6 -0
- data/resources/unicode_data/properties/Indic_Positional_Category/Top_And_Left_And_Right/value.dump +4 -0
- data/resources/unicode_data/properties/Indic_Positional_Category/Top_And_Right/value.dump +12 -0
- data/resources/unicode_data/properties/Indic_Positional_Category/Visual_Order_Left/value.dump +8 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Avagraha/value.dump +15 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Bindu/value.dump +0 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Consonant/value.dump +0 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Consonant_Dead/value.dump +4 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Consonant_Final/value.dump +13 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Consonant_Head_Letter/value.dump +3 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Consonant_Medial/value.dump +12 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Consonant_Placeholder/value.dump +0 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Consonant_Repha/value.dump +8 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Consonant_Subjoined/value.dump +11 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Modifying_Letter/value.dump +3 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Nukta/value.dump +18 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Register_Shifter/value.dump +3 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Tone_Letter/value.dump +5 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Tone_Mark/value.dump +17 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Virama/value.dump +41 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Visarga/value.dump +32 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Vowel/value.dump +6 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Vowel_Dependent/value.dump +112 -0
- data/resources/unicode_data/properties/Indic_Syllabic_Category/Vowel_Independent/value.dump +0 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/A/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/AE/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/B/value.dump +4 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/BB/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/BS/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/C/value.dump +4 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/D/value.dump +4 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/DD/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/E/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/EO/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/EU/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/G/value.dump +0 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/GG/value.dump +4 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/GS/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/H/value.dump +4 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/I/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/J/value.dump +4 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/JJ/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/K/value.dump +4 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/L/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/LB/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/LG/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/LH/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/LM/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/LP/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/LS/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/LT/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/M/value.dump +4 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/N/value.dump +4 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/NG/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/NH/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/NJ/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/O/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/OE/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/P/value.dump +4 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/R/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/S/value.dump +4 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/SS/value.dump +6 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/T/value.dump +4 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/U/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/WA/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/WAE/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/WE/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/WEO/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/WI/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/YA/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/YAE/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/YE/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/YEO/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/YI/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/YO/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/YU/value.dump +3 -0
- data/resources/unicode_data/properties/Jamo_Short_Name/value.dump +3 -0
- data/resources/unicode_data/properties/Join_Control/value.dump +3 -0
- data/resources/unicode_data/properties/Line_Break/AI/value.dump +0 -0
- data/resources/unicode_data/properties/Line_Break/AL/value.dump +0 -0
- data/resources/unicode_data/properties/Line_Break/B2/value.dump +4 -0
- data/resources/unicode_data/properties/Line_Break/BA/value.dump +0 -0
- data/resources/unicode_data/properties/Line_Break/BB/value.dump +15 -0
- data/resources/unicode_data/properties/Line_Break/BK/value.dump +4 -0
- data/resources/unicode_data/properties/Line_Break/CB/value.dump +3 -0
- data/resources/unicode_data/properties/Line_Break/CJ/value.dump +27 -0
- data/resources/unicode_data/properties/Line_Break/CL/value.dump +81 -0
- data/resources/unicode_data/properties/Line_Break/CM/value.dump +0 -0
- data/resources/unicode_data/properties/Line_Break/CP/value.dump +4 -0
- data/resources/unicode_data/properties/Line_Break/CR/value.dump +3 -0
- data/resources/unicode_data/properties/Line_Break/EX/value.dump +24 -0
- data/resources/unicode_data/properties/Line_Break/GL/value.dump +13 -0
- data/resources/unicode_data/properties/Line_Break/H2/value.dump +0 -0
- data/resources/unicode_data/properties/Line_Break/H3/value.dump +401 -0
- data/resources/unicode_data/properties/Line_Break/HL/value.dump +12 -0
- data/resources/unicode_data/properties/Line_Break/HY/value.dump +3 -0
- data/resources/unicode_data/properties/Line_Break/ID/value.dump +0 -0
- data/resources/unicode_data/properties/Line_Break/IN/value.dump +4 -0
- data/resources/unicode_data/properties/Line_Break/IS/value.dump +12 -0
- data/resources/unicode_data/properties/Line_Break/JL/value.dump +0 -0
- data/resources/unicode_data/properties/Line_Break/JT/value.dump +4 -0
- data/resources/unicode_data/properties/Line_Break/JV/value.dump +4 -0
- data/resources/unicode_data/properties/Line_Break/LF/value.dump +3 -0
- data/resources/unicode_data/properties/Line_Break/NL/value.dump +3 -0
- data/resources/unicode_data/properties/Line_Break/NS/value.dump +17 -0
- data/resources/unicode_data/properties/Line_Break/NU/value.dump +0 -0
- data/resources/unicode_data/properties/Line_Break/OP/value.dump +83 -0
- data/resources/unicode_data/properties/Line_Break/PO/value.dump +20 -0
- data/resources/unicode_data/properties/Line_Break/PR/value.dump +24 -0
- data/resources/unicode_data/properties/Line_Break/QU/value.dump +0 -0
- data/resources/unicode_data/properties/Line_Break/RI/value.dump +3 -0
- data/resources/unicode_data/properties/Line_Break/SA/value.dump +0 -0
- data/resources/unicode_data/properties/Line_Break/SG/value.dump +0 -0
- data/resources/unicode_data/properties/Line_Break/SP/value.dump +3 -0
- data/resources/unicode_data/properties/Line_Break/SY/value.dump +3 -0
- data/resources/unicode_data/properties/Line_Break/WJ/value.dump +4 -0
- data/resources/unicode_data/properties/Line_Break/XX/value.dump +0 -0
- data/resources/unicode_data/properties/Line_Break/ZW/value.dump +3 -0
- data/resources/unicode_data/properties/Logical_Order_Exception/value.dump +8 -0
- data/resources/unicode_data/properties/Lowercase/value.dump +0 -0
- data/resources/unicode_data/properties/Math/value.dump +0 -0
- data/resources/unicode_data/properties/Noncharacter_Code_Point/value.dump +22 -0
- data/resources/unicode_data/properties/Numeric_Type/0/value.dump +0 -0
- data/resources/unicode_data/properties/Numeric_Type/1/value.dump +69 -0
- data/resources/unicode_data/properties/Numeric_Type/2/value.dump +68 -0
- data/resources/unicode_data/properties/Numeric_Type/3/value.dump +68 -0
- data/resources/unicode_data/properties/Numeric_Type/4/value.dump +68 -0
- data/resources/unicode_data/properties/Numeric_Type/5/value.dump +65 -0
- data/resources/unicode_data/properties/Numeric_Type/6/value.dump +65 -0
- data/resources/unicode_data/properties/Numeric_Type/7/value.dump +65 -0
- data/resources/unicode_data/properties/Numeric_Type/8/value.dump +65 -0
- data/resources/unicode_data/properties/Numeric_Type/9/value.dump +67 -0
- data/resources/unicode_data/properties/Numeric_Type/value.dump +0 -0
- data/resources/unicode_data/properties/Other_Alphabetic/value.dump +0 -0
- data/resources/unicode_data/properties/Other_Default_Ignorable_Code_Point/value.dump +0 -0
- data/resources/unicode_data/properties/Other_Grapheme_Extend/value.dump +19 -0
- data/resources/unicode_data/properties/Other_ID_Continue/value.dump +6 -0
- data/resources/unicode_data/properties/Other_ID_Start/value.dump +5 -0
- data/resources/unicode_data/properties/Other_Lowercase/value.dump +20 -0
- data/resources/unicode_data/properties/Other_Math/value.dump +0 -0
- data/resources/unicode_data/properties/Other_Uppercase/value.dump +4 -0
- data/resources/unicode_data/properties/Pattern_Syntax/value.dump +0 -0
- data/resources/unicode_data/properties/Pattern_White_Space/value.dump +8 -0
- data/resources/unicode_data/properties/Quotation_Mark/value.dump +14 -0
- data/resources/unicode_data/properties/Radical/value.dump +0 -0
- data/resources/unicode_data/properties/STerm/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Arabic/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Armenian/value.dump +8 -0
- data/resources/unicode_data/properties/Script/Avestan/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Balinese/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Bamum/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Batak/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Bengali/value.dump +16 -0
- data/resources/unicode_data/properties/Script/Bopomofo/value.dump +5 -0
- data/resources/unicode_data/properties/Script/Brahmi/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Braille/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Buginese/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Buhid/value.dump +3 -0
- data/resources/unicode_data/properties/Script/Canadian_Aboriginal/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Carian/value.dump +3 -0
- data/resources/unicode_data/properties/Script/Chakma/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Cham/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Cherokee/value.dump +3 -0
- data/resources/unicode_data/properties/Script/Common/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Coptic/value.dump +5 -0
- data/resources/unicode_data/properties/Script/Cuneiform/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Cypriot/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Cyrillic/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Deseret/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Devanagari/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Egyptian_Hieroglyphs/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Ethiopic/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Georgian/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Glagolitic/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Gothic/value.dump +3 -0
- data/resources/unicode_data/properties/Script/Greek/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Gujarati/value.dump +40 -0
- data/resources/unicode_data/properties/Script/Gurmukhi/value.dump +50 -0
- data/resources/unicode_data/properties/Script/Han/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Hangul/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Hanunoo/value.dump +3 -0
- data/resources/unicode_data/properties/Script/Hebrew/value.dump +11 -0
- data/resources/unicode_data/properties/Script/Hiragana/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Imperial_Aramaic/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Inherited/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Inscriptional_Pahlavi/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Inscriptional_Parthian/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Javanese/value.dump +5 -0
- data/resources/unicode_data/properties/Script/Kaithi/value.dump +3 -0
- data/resources/unicode_data/properties/Script/Kannada/value.dump +16 -0
- data/resources/unicode_data/properties/Script/Katakana/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Kayah_Li/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Kharoshthi/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Khmer/value.dump +6 -0
- data/resources/unicode_data/properties/Script/Lao/value.dump +20 -0
- data/resources/unicode_data/properties/Script/Latin/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Lepcha/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Limbu/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Linear_B/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Lisu/value.dump +3 -0
- data/resources/unicode_data/properties/Script/Lycian/value.dump +3 -0
- data/resources/unicode_data/properties/Script/Lydian/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Malayalam/value.dump +13 -0
- data/resources/unicode_data/properties/Script/Mandaic/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Meetei_Mayek/value.dump +5 -0
- data/resources/unicode_data/properties/Script/Meroitic_Cursive/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Meroitic_Hieroglyphs/value.dump +3 -0
- data/resources/unicode_data/properties/Script/Miao/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Mongolian/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Myanmar/value.dump +0 -0
- data/resources/unicode_data/properties/Script/New_Tai_Lue/value.dump +6 -0
- data/resources/unicode_data/properties/Script/Nko/value.dump +3 -0
- data/resources/unicode_data/properties/Script/Ogham/value.dump +3 -0
- data/resources/unicode_data/properties/Script/Ol_Chiki/value.dump +3 -0
- data/resources/unicode_data/properties/Script/Old_Italic/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Old_Persian/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Old_South_Arabian/value.dump +5 -0
- data/resources/unicode_data/properties/Script/Old_Turkic/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Oriya/value.dump +16 -0
- data/resources/unicode_data/properties/Script/Osmanya/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Phags_Pa/value.dump +3 -0
- data/resources/unicode_data/properties/Script/Phoenician/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Rejang/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Runic/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Samaritan/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Saurashtra/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Sharada/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Shavian/value.dump +3 -0
- data/resources/unicode_data/properties/Script/Sinhala/value.dump +13 -0
- data/resources/unicode_data/properties/Script/Sora_Sompeng/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Sundanese/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Syloti_Nagri/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Syriac/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Tagalog/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Tagbanwa/value.dump +5 -0
- data/resources/unicode_data/properties/Script/Tai_Le/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Tai_Tham/value.dump +8 -0
- data/resources/unicode_data/properties/Script/Tai_Viet/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Takri/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Tamil/value.dump +18 -0
- data/resources/unicode_data/properties/Script/Telugu/value.dump +16 -0
- data/resources/unicode_data/properties/Script/Thaana/value.dump +3 -0
- data/resources/unicode_data/properties/Script/Thai/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Tibetan/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Tifinagh/value.dump +5 -0
- data/resources/unicode_data/properties/Script/Ugaritic/value.dump +4 -0
- data/resources/unicode_data/properties/Script/Vai/value.dump +0 -0
- data/resources/unicode_data/properties/Script/Yi/value.dump +0 -0
- data/resources/unicode_data/properties/Script_Extensions/Arab/value.dump +11 -0
- data/resources/unicode_data/properties/Script_Extensions/Armn/value.dump +3 -0
- data/resources/unicode_data/properties/Script_Extensions/Beng/value.dump +3 -0
- data/resources/unicode_data/properties/Script_Extensions/Bopo/value.dump +19 -0
- data/resources/unicode_data/properties/Script_Extensions/Bugi/value.dump +3 -0
- data/resources/unicode_data/properties/Script_Extensions/Buhd/value.dump +3 -0
- data/resources/unicode_data/properties/Script_Extensions/Cakm/value.dump +4 -0
- data/resources/unicode_data/properties/Script_Extensions/Cprt/value.dump +0 -0
- data/resources/unicode_data/properties/Script_Extensions/Cyrl/value.dump +3 -0
- data/resources/unicode_data/properties/Script_Extensions/Deva/value.dump +4 -0
- data/resources/unicode_data/properties/Script_Extensions/Geor/value.dump +3 -0
- data/resources/unicode_data/properties/Script_Extensions/Grek/value.dump +5 -0
- data/resources/unicode_data/properties/Script_Extensions/Gujr/value.dump +3 -0
- data/resources/unicode_data/properties/Script_Extensions/Guru/value.dump +4 -0
- data/resources/unicode_data/properties/Script_Extensions/Hang/value.dump +18 -0
- data/resources/unicode_data/properties/Script_Extensions/Hani/value.dump +21 -0
- data/resources/unicode_data/properties/Script_Extensions/Hano/value.dump +3 -0
- data/resources/unicode_data/properties/Script_Extensions/Hira/value.dump +24 -0
- data/resources/unicode_data/properties/Script_Extensions/Java/value.dump +3 -0
- data/resources/unicode_data/properties/Script_Extensions/Kana/value.dump +24 -0
- data/resources/unicode_data/properties/Script_Extensions/Kthi/value.dump +3 -0
- data/resources/unicode_data/properties/Script_Extensions/Latn/value.dump +5 -0
- data/resources/unicode_data/properties/Script_Extensions/Linb/value.dump +0 -0
- data/resources/unicode_data/properties/Script_Extensions/Mand/value.dump +3 -0
- data/resources/unicode_data/properties/Script_Extensions/Mong/value.dump +4 -0
- data/resources/unicode_data/properties/Script_Extensions/Mymr/value.dump +3 -0
- data/resources/unicode_data/properties/Script_Extensions/Orya/value.dump +3 -0
- data/resources/unicode_data/properties/Script_Extensions/Phag/value.dump +4 -0
- data/resources/unicode_data/properties/Script_Extensions/Sylo/value.dump +3 -0
- data/resources/unicode_data/properties/Script_Extensions/Syrc/value.dump +8 -0
- data/resources/unicode_data/properties/Script_Extensions/Tagb/value.dump +3 -0
- data/resources/unicode_data/properties/Script_Extensions/Takr/value.dump +3 -0
- data/resources/unicode_data/properties/Script_Extensions/Tale/value.dump +3 -0
- data/resources/unicode_data/properties/Script_Extensions/Tglg/value.dump +3 -0
- data/resources/unicode_data/properties/Script_Extensions/Thaa/value.dump +8 -0
- data/resources/unicode_data/properties/Script_Extensions/Yiii/value.dump +8 -0
- data/resources/unicode_data/properties/Sentence_Break/ATerm/value.dump +6 -0
- data/resources/unicode_data/properties/Sentence_Break/CR/value.dump +3 -0
- data/resources/unicode_data/properties/Sentence_Break/Close/value.dump +0 -0
- data/resources/unicode_data/properties/Sentence_Break/Extend/value.dump +0 -0
- data/resources/unicode_data/properties/Sentence_Break/Format/value.dump +0 -0
- data/resources/unicode_data/properties/Sentence_Break/LF/value.dump +3 -0
- data/resources/unicode_data/properties/Sentence_Break/Lower/value.dump +0 -0
- data/resources/unicode_data/properties/Sentence_Break/Numeric/value.dump +0 -0
- data/resources/unicode_data/properties/Sentence_Break/OLetter/value.dump +0 -0
- data/resources/unicode_data/properties/Sentence_Break/SContinue/value.dump +21 -0
- data/resources/unicode_data/properties/Sentence_Break/STerm/value.dump +0 -0
- data/resources/unicode_data/properties/Sentence_Break/Sep/value.dump +4 -0
- data/resources/unicode_data/properties/Sentence_Break/Sp/value.dump +0 -0
- data/resources/unicode_data/properties/Sentence_Break/Upper/value.dump +0 -0
- data/resources/unicode_data/properties/Soft_Dotted/value.dump +33 -0
- data/resources/unicode_data/properties/Terminal_Punctuation/value.dump +0 -0
- data/resources/unicode_data/properties/Unified_Ideograph/value.dump +0 -0
- data/resources/unicode_data/properties/Uppercase/value.dump +0 -0
- data/resources/unicode_data/properties/Variation_Selector/value.dump +0 -0
- data/resources/unicode_data/properties/White_Space/value.dump +0 -0
- data/resources/unicode_data/properties/Word_Break/ALetter/value.dump +0 -0
- data/resources/unicode_data/properties/Word_Break/CR/value.dump +3 -0
- data/resources/unicode_data/properties/Word_Break/Double_Quote/value.dump +3 -0
- data/resources/unicode_data/properties/Word_Break/Extend/value.dump +0 -0
- data/resources/unicode_data/properties/Word_Break/ExtendNumLet/value.dump +8 -0
- data/resources/unicode_data/properties/Word_Break/Format/value.dump +0 -0
- data/resources/unicode_data/properties/Word_Break/Hebrew_Letter/value.dump +12 -0
- data/resources/unicode_data/properties/Word_Break/Katakana/value.dump +0 -0
- data/resources/unicode_data/properties/Word_Break/LF/value.dump +3 -0
- data/resources/unicode_data/properties/Word_Break/MidLetter/value.dump +10 -0
- data/resources/unicode_data/properties/Word_Break/MidNum/value.dump +16 -0
- data/resources/unicode_data/properties/Word_Break/MidNumLet/value.dump +9 -0
- data/resources/unicode_data/properties/Word_Break/Newline/value.dump +5 -0
- data/resources/unicode_data/properties/Word_Break/Numeric/value.dump +0 -0
- data/resources/unicode_data/properties/Word_Break/Regional_Indicator/value.dump +3 -0
- data/resources/unicode_data/properties/Word_Break/Single_Quote/value.dump +3 -0
- data/resources/unicode_data/properties/XID_Continue/value.dump +0 -0
- data/resources/unicode_data/properties/XID_Start/value.dump +0 -0
- data/resources/unicode_data/property_aliases.yml +350 -0
- data/resources/unicode_data/property_value_aliases.yml +1829 -0
- data/spec/bidi/bidi_spec.rb +2 -2
- data/spec/collation/collation_spec.rb +1 -1
- data/spec/collation/collator_spec.rb +6 -6
- data/spec/collation/sort_key_builder_spec.rb +11 -11
- data/spec/collation/tailoring_spec.rb +1 -1
- data/spec/collation/trie_dumps_spec.rb +1 -1
- data/spec/data_readers/date_time_data_reader_spec.rb +1 -1
- data/spec/data_readers/number_data_reader_spec.rb +19 -6
- data/spec/formatters/calendars/datetime_formatter_spec.rb +1 -1
- data/spec/formatters/numbers/abbreviated/abbreviated_number_formatter_spec.rb +1 -1
- data/spec/formatters/numbers/abbreviated/long_decimal_formatter_spec.rb +5 -5
- data/spec/formatters/numbers/abbreviated/short_decimal_formatter_spec.rb +4 -4
- data/spec/formatters/numbers/currency_formatter_spec.rb +10 -10
- data/spec/formatters/numbers/decimal_formatter_spec.rb +2 -2
- data/spec/formatters/numbers/helpers/fraction_spec.rb +4 -4
- data/spec/formatters/numbers/helpers/integer_spec.rb +21 -16
- data/spec/formatters/numbers/number_formatter_spec.rb +9 -9
- data/spec/formatters/numbers/percent_formatter_spec.rb +2 -2
- data/spec/formatters/numbers/rbnf/rbnf_spec.rb +5 -5
- data/spec/formatters/plurals/plural_formatter_spec.rb +25 -25
- data/spec/formatters/plurals/rules_spec.rb +4 -4
- data/spec/localized/localized_date_spec.rb +25 -25
- data/spec/localized/localized_datetime_spec.rb +7 -7
- data/spec/localized/localized_hash_spec.rb +1 -1
- data/spec/localized/localized_number_spec.rb +23 -23
- data/spec/localized/localized_object_spec.rb +2 -2
- data/spec/localized/localized_string_spec.rb +43 -16
- data/spec/localized/localized_symbol_spec.rb +31 -4
- data/spec/localized/localized_time_spec.rb +3 -3
- data/spec/localized/localized_timespan_spec.rb +42 -42
- data/spec/normalization_spec.rb +4 -4
- data/spec/parsers/number_parser_spec.rb +10 -10
- data/spec/parsers/parser_spec.rb +19 -3
- data/spec/parsers/symbol_table_spec.rb +1 -1
- data/spec/parsers/unicode_regex/character_class_spec.rb +12 -0
- data/spec/parsers/unicode_regex/character_range_spec.rb +24 -5
- data/spec/parsers/unicode_regex/character_set_spec.rb +9 -1
- data/spec/parsers/unicode_regex_parser_spec.rb +1 -1
- data/spec/resources/loader_spec.rb +6 -6
- data/spec/{shared → segmentation}/break_iterator_spec.rb +45 -16
- data/spec/segmentation/parser_spec.rb +107 -0
- data/spec/segmentation/rule_set_spec.rb +102 -0
- data/spec/shared/calendar_spec.rb +30 -30
- data/spec/shared/caser_spec.rb +79 -0
- data/spec/shared/code_point_spec.rb +52 -151
- data/spec/shared/currencies_spec.rb +8 -8
- data/spec/shared/language_codes_spec.rb +13 -13
- data/spec/shared/likely_subtags_spec.rb +58 -0
- data/spec/shared/locale_spec.rb +211 -0
- data/spec/shared/numbers_spec.rb +4 -4
- data/spec/shared/postal_codes_spec.rb +24 -4
- data/spec/shared/properties_database_spec.rb +157 -0
- data/spec/shared/property_name_aliases_spec.rb +56 -0
- data/spec/shared/property_normalizer_spec.rb +64 -0
- data/spec/shared/property_set_spec.rb +218 -0
- data/spec/shared/property_value_aliases_spec.rb +58 -0
- data/spec/shared/territories_spec.rb +1 -1
- data/spec/shared/unicode_regex_spec.rb +35 -2
- data/spec/spec_helper.rb +3 -3
- data/spec/tokenizers/calendars/date_tokenizer_spec.rb +23 -23
- data/spec/tokenizers/calendars/datetime_tokenizer_spec.rb +18 -18
- data/spec/tokenizers/calendars/time_tokenizer_spec.rb +19 -19
- data/spec/tokenizers/composite_token_spec.rb +4 -4
- data/spec/tokenizers/numbers/number_tokenizer_spec.rb +16 -16
- data/spec/tokenizers/token_spec.rb +2 -2
- data/spec/tokenizers/unicode_regex/unicode_regex_tokenizer_spec.rb +94 -94
- data/spec/utils/file_system_trie_spec.rb +98 -0
- data/spec/utils/range_set_spec.rb +53 -1
- data/spec/utils/script_detector_spec.rb +58 -0
- data/spec/utils/yaml/yaml_spec.rb +22 -22
- data/spec/utils_spec.rb +21 -21
- metadata +832 -28
- data/lib/twitter_cldr/parsers/segmentation_parser.rb +0 -137
- data/lib/twitter_cldr/resources/canonical_compositions_updater.rb +0 -51
- data/lib/twitter_cldr/resources/composition_exclusions_importer.rb +0 -62
- data/lib/twitter_cldr/resources/normalization_quick_check_importer.rb +0 -73
- data/lib/twitter_cldr/resources/unicode_properties_importer.rb +0 -79
- data/lib/twitter_cldr/shared/break_iterator.rb +0 -213
- data/lib/twitter_cldr/tokenizers/segmentation/segmentation_tokenizer.rb +0 -39
- data/resources/shared/segments/tailorings/en.yml +0 -8
- data/resources/unicode_data/canonical_compositions.yml +0 -4925
- data/resources/unicode_data/composition_exclusions.yml +0 -297
- data/resources/unicode_data/hangul_blocks.yml +0 -21
- data/resources/unicode_data/indices/bidi_class.yml +0 -4572
- data/resources/unicode_data/indices/bidi_mirrored.yml +0 -3087
- data/resources/unicode_data/indices/category.yml +0 -10918
- data/resources/unicode_data/indices/keys.yml +0 -101
- data/resources/unicode_data/nfc_quick_check.yml +0 -293
- data/resources/unicode_data/nfd_quick_check.yml +0 -909
- data/resources/unicode_data/nfkc_quick_check.yml +0 -989
- data/resources/unicode_data/nfkd_quick_check.yml +0 -1537
- data/resources/unicode_data/properties/line_break.yml +0 -9269
- data/resources/unicode_data/properties/sentence_break.yml +0 -8067
- data/resources/unicode_data/properties/word_break.yml +0 -3001
- data/spec/parsers/segmentation_parser_spec.rb +0 -100
- data/spec/tokenizers/segmentation/segmentation_tokenizer_spec.rb +0 -40
@@ -1,137 +0,0 @@
|
|
1
|
-
# encoding: UTF-8
|
2
|
-
|
3
|
-
# Copyright 2012 Twitter, Inc
|
4
|
-
# http://www.apache.org/licenses/LICENSE-2.0
|
5
|
-
|
6
|
-
module TwitterCldr
|
7
|
-
module Parsers
|
8
|
-
|
9
|
-
class SegmentationParser < Parser
|
10
|
-
|
11
|
-
RuleMatchData = Struct.new(:text, :boundary_offset)
|
12
|
-
|
13
|
-
class Rule
|
14
|
-
attr_accessor :string, :id
|
15
|
-
end
|
16
|
-
|
17
|
-
class BreakRule < Rule
|
18
|
-
|
19
|
-
attr_reader :left, :right
|
20
|
-
|
21
|
-
def initialize(left, right)
|
22
|
-
@left = left
|
23
|
-
@right = right
|
24
|
-
end
|
25
|
-
|
26
|
-
def match(str)
|
27
|
-
if left && left_match = left.match(str)
|
28
|
-
match_pos = left_match.offset(0).last
|
29
|
-
|
30
|
-
if right
|
31
|
-
if right_match = right.match(str[match_pos..-1])
|
32
|
-
RuleMatchData.new(
|
33
|
-
left_match[0] + right_match[0],
|
34
|
-
match_pos
|
35
|
-
)
|
36
|
-
end
|
37
|
-
else
|
38
|
-
RuleMatchData.new(str, str.size)
|
39
|
-
end
|
40
|
-
end
|
41
|
-
end
|
42
|
-
|
43
|
-
def boundary_symbol
|
44
|
-
:break
|
45
|
-
end
|
46
|
-
|
47
|
-
end
|
48
|
-
|
49
|
-
class NoBreakRule < Rule
|
50
|
-
|
51
|
-
attr_reader :regex
|
52
|
-
|
53
|
-
def initialize(regex)
|
54
|
-
@regex = regex
|
55
|
-
end
|
56
|
-
|
57
|
-
def match(str)
|
58
|
-
if match = regex.match(str)
|
59
|
-
RuleMatchData.new(match[0], match.offset(0).last)
|
60
|
-
end
|
61
|
-
end
|
62
|
-
|
63
|
-
def boundary_symbol
|
64
|
-
:no_break
|
65
|
-
end
|
66
|
-
|
67
|
-
end
|
68
|
-
|
69
|
-
private
|
70
|
-
|
71
|
-
def do_parse(options)
|
72
|
-
regex_token_lists = []
|
73
|
-
current_regex_tokens = []
|
74
|
-
boundary_symbol = nil
|
75
|
-
|
76
|
-
while current_token
|
77
|
-
case current_token.type
|
78
|
-
when :break, :no_break
|
79
|
-
boundary_symbol = current_token.type
|
80
|
-
regex_token_lists << current_regex_tokens
|
81
|
-
current_regex_tokens = []
|
82
|
-
else
|
83
|
-
current_regex_tokens << current_token
|
84
|
-
end
|
85
|
-
|
86
|
-
next_token(current_token.type)
|
87
|
-
end
|
88
|
-
|
89
|
-
regex_token_lists << current_regex_tokens
|
90
|
-
|
91
|
-
case boundary_symbol
|
92
|
-
when :break
|
93
|
-
BreakRule.new(
|
94
|
-
parse_regex(add_anchors(regex_token_lists[0]), options),
|
95
|
-
parse_regex(add_anchors(regex_token_lists[1]), options)
|
96
|
-
)
|
97
|
-
when :no_break
|
98
|
-
NoBreakRule.new(
|
99
|
-
parse_regex(add_anchors(regex_token_lists.flatten), options)
|
100
|
-
)
|
101
|
-
end
|
102
|
-
end
|
103
|
-
|
104
|
-
# only find matches from the beginning of the current string
|
105
|
-
def add_anchors(token_list)
|
106
|
-
token_list.insert(0, begin_token)
|
107
|
-
end
|
108
|
-
|
109
|
-
def begin_token
|
110
|
-
self.class.begin_token
|
111
|
-
end
|
112
|
-
|
113
|
-
def self.begin_token
|
114
|
-
@begin_token ||= TwitterCldr::Tokenizers::Token.new(
|
115
|
-
:type => :special_char, :value => "\\A"
|
116
|
-
)
|
117
|
-
end
|
118
|
-
|
119
|
-
def parse_regex(tokens, options)
|
120
|
-
unless tokens.empty?
|
121
|
-
TwitterCldr::Shared::UnicodeRegex.new(
|
122
|
-
regex_parser.parse(tokens, options)
|
123
|
-
)
|
124
|
-
end
|
125
|
-
end
|
126
|
-
|
127
|
-
def regex_parser
|
128
|
-
self.class.regex_parser
|
129
|
-
end
|
130
|
-
|
131
|
-
def self.regex_parser
|
132
|
-
@regex_parser ||= UnicodeRegexParser.new
|
133
|
-
end
|
134
|
-
|
135
|
-
end
|
136
|
-
end
|
137
|
-
end
|
@@ -1,51 +0,0 @@
|
|
1
|
-
# encoding: UTF-8
|
2
|
-
|
3
|
-
# Copyright 2012 Twitter, Inc
|
4
|
-
# http://www.apache.org/licenses/LICENSE-2.0
|
5
|
-
|
6
|
-
module TwitterCldr
|
7
|
-
module Resources
|
8
|
-
|
9
|
-
class CanonicalCompositionsUpdater
|
10
|
-
|
11
|
-
CODE_POINT_MAX = 0x10FFFF
|
12
|
-
|
13
|
-
# Arguments:
|
14
|
-
#
|
15
|
-
# output_path - output directory for generated YAML file
|
16
|
-
#
|
17
|
-
def initialize(output_path)
|
18
|
-
@output_path = output_path
|
19
|
-
end
|
20
|
-
|
21
|
-
def update
|
22
|
-
File.open(File.join(@output_path, 'canonical_compositions.yml'), 'w') do |output|
|
23
|
-
YAML.dump(generate_compositions, output)
|
24
|
-
end
|
25
|
-
end
|
26
|
-
|
27
|
-
private
|
28
|
-
|
29
|
-
def generate_compositions
|
30
|
-
(1..CODE_POINT_MAX).inject({}) do |memo, code_point|
|
31
|
-
code_point_data = TwitterCldr::Shared::CodePoint.find(code_point)
|
32
|
-
|
33
|
-
if code_point_data && !code_point_data.compatibility_decomposition? && code_point_data.decomposition && !code_point_data.decomposition.empty?
|
34
|
-
memo[code_point_data.decomposition] = code_point
|
35
|
-
end
|
36
|
-
|
37
|
-
log_progress(code_point, memo.size)
|
38
|
-
|
39
|
-
memo
|
40
|
-
end
|
41
|
-
end
|
42
|
-
|
43
|
-
def log_progress(code_point, compositions_count)
|
44
|
-
$stdout.write("\r#{(100.0 * code_point / CODE_POINT_MAX).round}% complete, found #{compositions_count} canonical compositions")
|
45
|
-
$stdout.write("\n") if code_point == CODE_POINT_MAX
|
46
|
-
end
|
47
|
-
|
48
|
-
end
|
49
|
-
|
50
|
-
end
|
51
|
-
end
|
@@ -1,62 +0,0 @@
|
|
1
|
-
# encoding: UTF-8
|
2
|
-
|
3
|
-
# Copyright 2012 Twitter, Inc
|
4
|
-
# http://www.apache.org/licenses/LICENSE-2.0
|
5
|
-
|
6
|
-
require 'twitter_cldr/resources/download'
|
7
|
-
|
8
|
-
module TwitterCldr
|
9
|
-
module Resources
|
10
|
-
|
11
|
-
class CompositionExclusionsImporter
|
12
|
-
|
13
|
-
COMPOSITION_EXCLUSIONS_URL = 'http://www.unicode.org/Public/6.1.0/ucd/DerivedNormalizationProps.txt'
|
14
|
-
COMPOSITION_EXCLUSION_REGEXP = /^([0-9A-F]+)(?:\.\.([0-9A-F]+))?\s+; Full_Composition_Exclusion #.*$/
|
15
|
-
TOTAL_CODE_POINTS_REGEXP = /^# Total code points: (\d+)$/
|
16
|
-
|
17
|
-
# Arguments:
|
18
|
-
#
|
19
|
-
# input_path - path to DerivedNormalizationProps.txt file
|
20
|
-
# output_path - output directory for generated YAML file
|
21
|
-
#
|
22
|
-
def initialize(input_path, output_path)
|
23
|
-
@input_path = input_path
|
24
|
-
@output_path = output_path
|
25
|
-
end
|
26
|
-
|
27
|
-
def import
|
28
|
-
File.open(File.join(@output_path, 'composition_exclusions.yml'), 'w') do |output|
|
29
|
-
YAML.dump(generate_composition_exclusions, output)
|
30
|
-
end
|
31
|
-
end
|
32
|
-
|
33
|
-
private
|
34
|
-
|
35
|
-
def generate_composition_exclusions
|
36
|
-
data = File.open(composition_exclusions_file) { |file| file.read }
|
37
|
-
start_pos = data.index("# Derived Property: Full_Composition_Exclusion")
|
38
|
-
end_pos = data.index(/^#\s=*$/, start_pos)
|
39
|
-
data = data[start_pos..end_pos].split("\n")
|
40
|
-
|
41
|
-
expected_code_points_count = nil
|
42
|
-
|
43
|
-
result = data.inject([]) do |memo, line|
|
44
|
-
memo << ($1.hex..($2 || $1).hex) if line =~ COMPOSITION_EXCLUSION_REGEXP
|
45
|
-
expected_code_points_count = $1.to_i if line =~ TOTAL_CODE_POINTS_REGEXP
|
46
|
-
memo
|
47
|
-
end
|
48
|
-
|
49
|
-
raise "Expected number of code points was not found." unless expected_code_points_count
|
50
|
-
code_points_count = result.map(&:count).inject(:+)
|
51
|
-
raise "Unexpected number of code points: expected - #{expected_code_points_count}, got - #{code_points_count}." unless code_points_count == expected_code_points_count
|
52
|
-
|
53
|
-
result
|
54
|
-
end
|
55
|
-
|
56
|
-
def composition_exclusions_file
|
57
|
-
TwitterCldr::Resources.download_if_necessary(@input_path, COMPOSITION_EXCLUSIONS_URL)
|
58
|
-
end
|
59
|
-
|
60
|
-
end
|
61
|
-
end
|
62
|
-
end
|
@@ -1,73 +0,0 @@
|
|
1
|
-
# encoding: UTF-8
|
2
|
-
|
3
|
-
# Copyright 2012 Twitter, Inc
|
4
|
-
# http://www.apache.org/licenses/LICENSE-2.0
|
5
|
-
|
6
|
-
require 'twitter_cldr/resources/download'
|
7
|
-
|
8
|
-
module TwitterCldr
|
9
|
-
module Resources
|
10
|
-
|
11
|
-
class NormalizationQuickCheckImporter
|
12
|
-
|
13
|
-
PROPS_FILE_URL = "ftp://ftp.unicode.org/Public/UNIDATA/DerivedNormalizationProps.txt"
|
14
|
-
|
15
|
-
# Arguments:
|
16
|
-
#
|
17
|
-
# input_path - path to a directory containing DerivedNormalizationProps.txt
|
18
|
-
# output_path - output directory for imported YAML files
|
19
|
-
#
|
20
|
-
def initialize(input_path, output_path)
|
21
|
-
@input_path = input_path
|
22
|
-
@output_path = output_path
|
23
|
-
end
|
24
|
-
|
25
|
-
def import
|
26
|
-
parse_props_file.each_pair do |algorithm, code_point_list|
|
27
|
-
File.open(File.join(@output_path, "#{algorithm.downcase}_quick_check.yml"), "w+") do |f|
|
28
|
-
f.write(YAML.dump(TwitterCldr::Utils::RangeSet.rangify(code_point_list)))
|
29
|
-
end
|
30
|
-
end
|
31
|
-
end
|
32
|
-
|
33
|
-
private
|
34
|
-
|
35
|
-
def parse_props_file
|
36
|
-
check_table = {}
|
37
|
-
cur_type = nil
|
38
|
-
|
39
|
-
File.open(props_file) do |input|
|
40
|
-
input.each_line do |line|
|
41
|
-
cur_type = nil if line =~ /=Maybe/
|
42
|
-
type = line.scan(/#\s*Property:\s*(NF[KDC]+)_Quick_Check/).flatten
|
43
|
-
|
44
|
-
if type.size > 0
|
45
|
-
cur_type = type.first
|
46
|
-
check_table[cur_type] = []
|
47
|
-
end
|
48
|
-
|
49
|
-
if check_table.size > 0 && line[0...1] != "#" && !line.strip.empty? && cur_type
|
50
|
-
start, finish = line.scan(/(\h+(\.\.\h+)?)/).first.first.split("..").map { |num| num.to_i(16) }
|
51
|
-
|
52
|
-
if finish
|
53
|
-
check_table[cur_type] += (start..finish).to_a
|
54
|
-
else
|
55
|
-
check_table[cur_type] << start
|
56
|
-
end
|
57
|
-
end
|
58
|
-
|
59
|
-
break if line =~ /={5,}/ && check_table.size >= 4 && check_table.all? { |key, val| val.size > 0 }
|
60
|
-
end
|
61
|
-
end
|
62
|
-
|
63
|
-
check_table
|
64
|
-
end
|
65
|
-
|
66
|
-
def props_file
|
67
|
-
TwitterCldr::Resources.download_if_necessary(File.join(@input_path, 'DerivedNormalizationProps.txt'), PROPS_FILE_URL)
|
68
|
-
end
|
69
|
-
|
70
|
-
end
|
71
|
-
|
72
|
-
end
|
73
|
-
end
|
@@ -1,79 +0,0 @@
|
|
1
|
-
# encoding: UTF-8
|
2
|
-
|
3
|
-
# Copyright 2012 Twitter, Inc
|
4
|
-
# http://www.apache.org/licenses/LICENSE-2.0
|
5
|
-
|
6
|
-
require 'twitter_cldr/resources/download'
|
7
|
-
|
8
|
-
module TwitterCldr
|
9
|
-
module Resources
|
10
|
-
|
11
|
-
class UnicodePropertiesImporter < UnicodeImporter
|
12
|
-
|
13
|
-
PROPERTIES_BASE_URL = 'ftp://ftp.unicode.org/Public/UCD/latest/ucd'
|
14
|
-
PROPERTIES = [
|
15
|
-
"auxiliary/SentenceBreakProperty",
|
16
|
-
"auxiliary/WordBreakProperty",
|
17
|
-
"LineBreak"
|
18
|
-
]
|
19
|
-
|
20
|
-
# Arguments:
|
21
|
-
#
|
22
|
-
# input_path - path to a directory containing the various property files
|
23
|
-
# output_path - output directory for imported YAML files
|
24
|
-
#
|
25
|
-
def initialize(input_path, output_path)
|
26
|
-
@input_path = input_path
|
27
|
-
@output_path = output_path
|
28
|
-
end
|
29
|
-
|
30
|
-
def import
|
31
|
-
FileUtils.mkdir_p(@output_path)
|
32
|
-
|
33
|
-
PROPERTIES.each do |property|
|
34
|
-
input_file = property_data_file("#{property}.txt")
|
35
|
-
output_file = File.join(@output_path, "#{fix_name(property)}.yml")
|
36
|
-
|
37
|
-
File.open(output_file, "w+") do |f|
|
38
|
-
f.write(
|
39
|
-
YAML.dump(
|
40
|
-
parse_standard_file(input_file).inject({}) do |ret, data|
|
41
|
-
name = data[1].strip.to_sym
|
42
|
-
ret[name] ||= []
|
43
|
-
ret[name] += expand_range(data[0])
|
44
|
-
ret
|
45
|
-
end.inject({}) do |ret, (key, data)|
|
46
|
-
ret[key] = TwitterCldr::Utils::RangeSet.rangify(data)
|
47
|
-
ret
|
48
|
-
end
|
49
|
-
)
|
50
|
-
)
|
51
|
-
end
|
52
|
-
end
|
53
|
-
end
|
54
|
-
|
55
|
-
private
|
56
|
-
|
57
|
-
def fix_name(str)
|
58
|
-
underscore(File.basename(str.gsub("Property", "")))
|
59
|
-
end
|
60
|
-
|
61
|
-
def underscore(str)
|
62
|
-
str.gsub(/([a-z])([A-Z])/, '\1_\2').downcase
|
63
|
-
end
|
64
|
-
|
65
|
-
def expand_range(str)
|
66
|
-
initial, final = str.split("..")
|
67
|
-
(initial.to_i(16)..(final || initial).to_i(16)).to_a
|
68
|
-
end
|
69
|
-
|
70
|
-
def property_data_file(file)
|
71
|
-
TwitterCldr::Resources.download_if_necessary(
|
72
|
-
File.join(@input_path, File.basename(file)), File.join(PROPERTIES_BASE_URL, file)
|
73
|
-
)
|
74
|
-
end
|
75
|
-
|
76
|
-
end
|
77
|
-
|
78
|
-
end
|
79
|
-
end
|
@@ -1,213 +0,0 @@
|
|
1
|
-
# encoding: UTF-8
|
2
|
-
|
3
|
-
# Copyright 2012 Twitter, Inc
|
4
|
-
# http://www.apache.org/licenses/LICENSE-2.0
|
5
|
-
|
6
|
-
module TwitterCldr
|
7
|
-
module Shared
|
8
|
-
class BreakIterator
|
9
|
-
|
10
|
-
attr_reader :locale, :use_uli_exceptions
|
11
|
-
|
12
|
-
def initialize(locale = TwitterCldr.locale, options = {})
|
13
|
-
@use_uli_exceptions = !!options.fetch(:use_uli_exceptions, true)
|
14
|
-
@locale = locale
|
15
|
-
end
|
16
|
-
|
17
|
-
def each_sentence(str, &block)
|
18
|
-
each_boundary(str, "sentence", &block)
|
19
|
-
end
|
20
|
-
|
21
|
-
def each_word(str, &block)
|
22
|
-
raise NotImplementedError.new("Word segmentation is not currently supported.")
|
23
|
-
end
|
24
|
-
|
25
|
-
def each_line(str, &block)
|
26
|
-
raise NotImplementedError.new("Line segmentation is not currently supported.")
|
27
|
-
end
|
28
|
-
|
29
|
-
private
|
30
|
-
|
31
|
-
def boundary_name_for(str)
|
32
|
-
str.gsub(/(?:^|\_)([A-Za-z])/) { |s| $1.upcase } + "Break"
|
33
|
-
end
|
34
|
-
|
35
|
-
def each_boundary(str, boundary_type)
|
36
|
-
if block_given?
|
37
|
-
rules = compile_rules_for(locale, boundary_type)
|
38
|
-
match = nil
|
39
|
-
last_offset = 0
|
40
|
-
current_position = 0
|
41
|
-
search_str = str.dup
|
42
|
-
|
43
|
-
until search_str.size == 0
|
44
|
-
rule = rules.find { |rule| match = rule.match(search_str) }
|
45
|
-
|
46
|
-
if rule.boundary_symbol == :break
|
47
|
-
break_offset = current_position + match.boundary_offset
|
48
|
-
yield str[last_offset...break_offset]
|
49
|
-
last_offset = break_offset
|
50
|
-
end
|
51
|
-
|
52
|
-
search_str = search_str[match.boundary_offset..-1]
|
53
|
-
current_position += match.boundary_offset
|
54
|
-
end
|
55
|
-
|
56
|
-
if last_offset < (str.size - 1)
|
57
|
-
yield str[last_offset..-1]
|
58
|
-
end
|
59
|
-
else
|
60
|
-
to_enum(__method__, str, boundary_type)
|
61
|
-
end
|
62
|
-
end
|
63
|
-
|
64
|
-
# See the comment above exceptions_for. Basically, we only support exceptions
|
65
|
-
# for the "sentence" boundary type since the ULI JSON data doesn't distinguish
|
66
|
-
# between boundary types.
|
67
|
-
def compile_exception_rule_for(locale, boundary_type, boundary_name)
|
68
|
-
if boundary_type == "sentence"
|
69
|
-
cache_key = TwitterCldr::Utils.compute_cache_key(locale, boundary_type)
|
70
|
-
self.class.exceptions_cache[cache_key] ||= begin
|
71
|
-
exceptions = exceptions_for(locale, boundary_name)
|
72
|
-
regex_contents = exceptions.map { |exc| Regexp.escape(exc) }.join("|")
|
73
|
-
segmentation_parser.parse(
|
74
|
-
segmentation_tokenizer.tokenize("(?:#{regex_contents}) ×")
|
75
|
-
)
|
76
|
-
end
|
77
|
-
end
|
78
|
-
end
|
79
|
-
|
80
|
-
def self.exceptions_cache
|
81
|
-
@exceptions_cache ||= {}
|
82
|
-
end
|
83
|
-
|
84
|
-
# Grabs rules from segment_root, applies custom tailorings (our own, NOT from CLDR),
|
85
|
-
# and optionally integrates ULI exceptions.
|
86
|
-
def compile_rules_for(locale, boundary_type)
|
87
|
-
rules = self.class.rule_cache[boundary_type] ||= begin
|
88
|
-
boundary_name = boundary_name_for(boundary_type)
|
89
|
-
boundary_data = resource_for(boundary_name)
|
90
|
-
symbol_table = symbol_table_for(boundary_data)
|
91
|
-
root_rules = rules_for(boundary_data, symbol_table)
|
92
|
-
|
93
|
-
tailoring_boundary_data = tailoring_resource_for(locale, boundary_name)
|
94
|
-
tailoring_rules = rules_for(tailoring_boundary_data, symbol_table)
|
95
|
-
merge_rules(root_rules, tailoring_rules)
|
96
|
-
end
|
97
|
-
|
98
|
-
if use_uli_exceptions
|
99
|
-
exception_rule = compile_exception_rule_for(locale, boundary_type, boundary_name)
|
100
|
-
rules = rules.dup # avoid modifying the cached rules
|
101
|
-
rules.insert(0, exception_rule)
|
102
|
-
end
|
103
|
-
|
104
|
-
rules
|
105
|
-
end
|
106
|
-
|
107
|
-
# replaces ruleset1's rules with rules with the same id from ruleset2
|
108
|
-
def merge_rules(ruleset1, ruleset2)
|
109
|
-
result = ruleset1.dup
|
110
|
-
ruleset2.each do |new_rule|
|
111
|
-
if existing_idx = result.find_index { |rule| rule.id == new_rule.id }
|
112
|
-
result[existing_idx] = new_rule
|
113
|
-
end
|
114
|
-
end
|
115
|
-
result
|
116
|
-
end
|
117
|
-
|
118
|
-
def self.rule_cache
|
119
|
-
@rule_cache ||= {}
|
120
|
-
end
|
121
|
-
|
122
|
-
def symbol_table_for(boundary_data)
|
123
|
-
table = TwitterCldr::Parsers::SymbolTable.new
|
124
|
-
boundary_data[:variables].each do |variable|
|
125
|
-
id = variable[:id].to_s
|
126
|
-
tokens = segmentation_tokenizer.tokenize(variable[:value])
|
127
|
-
# note: variables can be redefined (add replaces if key already exists)
|
128
|
-
table.add(id, resolve_symbols(tokens, table))
|
129
|
-
end
|
130
|
-
table
|
131
|
-
end
|
132
|
-
|
133
|
-
def resolve_symbols(tokens, symbol_table)
|
134
|
-
tokens.inject([]) do |ret, token|
|
135
|
-
if token.type == :variable
|
136
|
-
ret += symbol_table.fetch(token.value)
|
137
|
-
else
|
138
|
-
ret << token
|
139
|
-
end
|
140
|
-
ret
|
141
|
-
end
|
142
|
-
end
|
143
|
-
|
144
|
-
def rules_for(boundary_data, symbol_table)
|
145
|
-
boundary_data[:rules].map do |rule|
|
146
|
-
r = segmentation_parser.parse(
|
147
|
-
segmentation_tokenizer.tokenize(rule[:value]), {
|
148
|
-
:symbol_table => symbol_table
|
149
|
-
}
|
150
|
-
)
|
151
|
-
|
152
|
-
r.string = rule[:value]
|
153
|
-
r.id = rule[:id]
|
154
|
-
r
|
155
|
-
end
|
156
|
-
end
|
157
|
-
|
158
|
-
def self.segmentation_tokenizer
|
159
|
-
@segmentation_tokenizer ||= TwitterCldr::Tokenizers::SegmentationTokenizer.new
|
160
|
-
end
|
161
|
-
|
162
|
-
def segmentation_tokenizer
|
163
|
-
self.class.segmentation_tokenizer
|
164
|
-
end
|
165
|
-
|
166
|
-
def self.segmentation_parser
|
167
|
-
@segmentation_parser ||= TwitterCldr::Parsers::SegmentationParser.new
|
168
|
-
end
|
169
|
-
|
170
|
-
def segmentation_parser
|
171
|
-
self.class.segmentation_parser
|
172
|
-
end
|
173
|
-
|
174
|
-
def resource_for(boundary_name)
|
175
|
-
self.class.root_resource[:segments][boundary_name.to_sym]
|
176
|
-
end
|
177
|
-
|
178
|
-
def tailoring_resource_for(locale, boundary_name)
|
179
|
-
cache_key = TwitterCldr::Utils.compute_cache_key(locale, boundary_name)
|
180
|
-
self.class.tailoring_resource_cache[cache_key] ||= begin
|
181
|
-
res = TwitterCldr.get_resource("shared", "segments", "tailorings", locale)
|
182
|
-
res[locale][:segments][boundary_name.to_sym]
|
183
|
-
end
|
184
|
-
end
|
185
|
-
|
186
|
-
def self.tailoring_resource_cache
|
187
|
-
@tailoring_resource_cache ||= {}
|
188
|
-
end
|
189
|
-
|
190
|
-
def self.root_resource
|
191
|
-
@root_resource ||= TwitterCldr.get_resource("shared", "segments", "segments_root")
|
192
|
-
end
|
193
|
-
|
194
|
-
# The boundary_name param is not currently used since the ULI JSON resource that
|
195
|
-
# exceptions are generated from does not distinguish between boundary types. The
|
196
|
-
# XML version does, however, so the JSON will hopefully catch up at some point and
|
197
|
-
# we can make use of this second parameter. For the time being, compile_exception_rule_for
|
198
|
-
# (which calls this function) assumes a "sentence" boundary type.
|
199
|
-
def exceptions_for(locale, boundary_name)
|
200
|
-
self.class.exceptions_resource_cache[locale] ||= begin
|
201
|
-
TwitterCldr.get_resource("uli", "segments", locale)[locale][:exceptions]
|
202
|
-
rescue ArgumentError
|
203
|
-
[]
|
204
|
-
end
|
205
|
-
end
|
206
|
-
|
207
|
-
def self.exceptions_resource_cache
|
208
|
-
@exceptions_resource_cache ||= {}
|
209
|
-
end
|
210
|
-
|
211
|
-
end
|
212
|
-
end
|
213
|
-
end
|
@@ -1,39 +0,0 @@
|
|
1
|
-
# encoding: UTF-8
|
2
|
-
|
3
|
-
# Copyright 2012 Twitter, Inc
|
4
|
-
# http://www.apache.org/licenses/LICENSE-2.0
|
5
|
-
|
6
|
-
module TwitterCldr
|
7
|
-
module Tokenizers
|
8
|
-
class SegmentationTokenizer
|
9
|
-
|
10
|
-
def tokenize(pattern)
|
11
|
-
# according to the spec, whitespace should be ignored
|
12
|
-
tokenizer.tokenize(pattern).reject do |token|
|
13
|
-
token.value.strip.empty?
|
14
|
-
end
|
15
|
-
end
|
16
|
-
|
17
|
-
private
|
18
|
-
|
19
|
-
def tokenizer
|
20
|
-
@tokenizer ||= begin
|
21
|
-
recognizers = [
|
22
|
-
TokenRecognizer.new(:break, /\303\267/u) do |val| # ÷ character
|
23
|
-
val.strip
|
24
|
-
end,
|
25
|
-
|
26
|
-
TokenRecognizer.new(:no_break, /\303\227/u) do |val| # × character
|
27
|
-
val.strip
|
28
|
-
end
|
29
|
-
]
|
30
|
-
|
31
|
-
ur_tokenizer = UnicodeRegexTokenizer.new
|
32
|
-
ur_tokenizer.insert_before(:string, *recognizers)
|
33
|
-
ur_tokenizer
|
34
|
-
end
|
35
|
-
end
|
36
|
-
|
37
|
-
end
|
38
|
-
end
|
39
|
-
end
|