diff --git a/Jenkinsfile b/Jenkinsfile index a7cd82c19..90d2db4c7 100644 --- a/Jenkinsfile +++ b/Jenkinsfile @@ -10,9 +10,9 @@ pipeline { disableConcurrentBuilds(abortPrevious: true) } environment { - AR_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/04-24-24-0' + AR_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/06-11-26-0' DE_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/10-23-24-0' - EN_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/06-04-26-3' + EN_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/06-11-26-1' ES_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/09-25-24-0' ES_EN_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/08-30-24-0' HI_EN_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/06-04-26-4' @@ -27,9 +27,9 @@ pipeline { HE_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/09-24-25-0' HY_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/03-12-24-0' MR_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/03-12-24-1' - JA_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/10-17-24-1' - HI_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/09-21-26-0' - KO_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/06-04-25-6' + JA_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/07-17-26-0' + KO_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/09-15-26-0' + HI_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/09-28-26-0' DEFAULT_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/06-08-23-0' } stages { @@ -332,7 +332,7 @@ pipeline { } } } - stage('L0: Create JA ITN Grammars') { + stage('L0: Create JA TN/ITN Grammars') { when { anyOf { branch 'main' @@ -348,6 +348,11 @@ pipeline { sh 'CUDA_VISIBLE_DEVICES="" python nemo_text_processing/inverse_text_normalization/inverse_normalize.py --lang=ja --text="100" --cache_dir ${JA_TN_CACHE}' } } + stage('L0: JA TN grammars') { + steps { + sh 'CUDA_VISIBLE_DEVICES="" python nemo_text_processing/text_normalization/normalize.py --lang=ja --text="100" --cache_dir ${JA_TN_CACHE}' + } + } } } stage('L0: Create KO TN/ITN Grammars') { @@ -481,7 +486,7 @@ pipeline { } } - stage('L2: EN Sparrowhawk Tests') { + stage('L2: EN Sparrowhawk Tests') { when { anyOf { branch 'main' diff --git a/nemo_text_processing/inverse_text_normalization/ja/data/time_hours.tsv b/nemo_text_processing/inverse_text_normalization/ja/data/time_hours.tsv index 8a5eced19..1957b95ad 100644 --- a/nemo_text_processing/inverse_text_normalization/ja/data/time_hours.tsv +++ b/nemo_text_processing/inverse_text_normalization/ja/data/time_hours.tsv @@ -22,4 +22,4 @@ 二十二 22 二十三 23 二十四 24 -零 0 \ No newline at end of file +ゼロ 0 \ No newline at end of file diff --git a/nemo_text_processing/inverse_text_normalization/ja/taggers/fraction.py b/nemo_text_processing/inverse_text_normalization/ja/taggers/fraction.py index 0ced0c679..10882ae99 100644 --- a/nemo_text_processing/inverse_text_normalization/ja/taggers/fraction.py +++ b/nemo_text_processing/inverse_text_normalization/ja/taggers/fraction.py @@ -26,7 +26,7 @@ def __init__(self, cardinal: GraphFst, decimal: GraphFst): e.g., 四分の三 -> fraction { denominator: "4" numerator: "3" } 一と四分の三 -> fraction { integer: "1" denominator: "4" numerator: "3" } - 一荷四分の三 -> fraction { integer: "1" denominator: "4" numerator: "3" } + 一と四分の三 -> fraction { integer: "1" denominator: "4" numerator: "3" } ルート三分の一 -> fraction { denominator: "√3" numerator: "1" } 一点六五分の五十 -> fraction { denominator: "1.65" numerator: "50" } 二ルート六分の三 -> -> fraction { denominator: "2√6 " numerator: "3" } @@ -40,7 +40,7 @@ def __init__(self, cardinal: GraphFst, decimal: GraphFst): pynutil.delete("分の") | pynutil.delete(" 分 の ") | pynutil.delete("分 の ") | pynutil.delete("分 の") ) - integer_word = pynutil.delete("と") | pynutil.delete("荷") + integer_word = pynutil.delete("と") root_word = pynini.accep("√") | pynini.cross("ルート", "√") graph_sign = ( diff --git a/nemo_text_processing/text_normalization/ar/taggers/money.py b/nemo_text_processing/text_normalization/ar/taggers/money.py index 925fa348e..b809354e4 100644 --- a/nemo_text_processing/text_normalization/ar/taggers/money.py +++ b/nemo_text_processing/text_normalization/ar/taggers/money.py @@ -80,14 +80,14 @@ def __init__(self, cardinal: GraphFst, deterministic: bool = True): pynutil.insert("integer_part: \"") + ((NEMO_SIGMA - "1") @ cardinal_graph) + pynutil.insert("\"") ) - graph_integer_only = graph_maj_singular + insert_space + graph_integer_one - graph_integer_only |= graph_maj_plural + insert_space + graph_integer + currency_first = pynutil.insert(' morphosyntactic_features: "currency_first"') + # Currency-first tagging for exactly one major unit (e.g. $1 -> دولار واحد). + graph_integer_one_unit = graph_maj_singular + insert_space + graph_integer_one + currency_first # For local currency "9د.ك" graph_integer_only_ar = graph_integer + insert_space + graph_ar_cur - # graph_decimal_ar = graph_decimal_final + insert_space + graph_ar_cur - graph = (graph_integer_only + optional_delete_fractional_zeros) | graph_integer_only_ar + graph = (graph_integer_one_unit + optional_delete_fractional_zeros) | graph_integer_only_ar # remove trailing zeros of non zero number in the first 2 digits and fill up to 2 digits # e.g. 2000 -> 20, 0200->02, 01 -> 01, 10 -> 10 @@ -112,9 +112,12 @@ def __init__(self, cardinal: GraphFst, deterministic: bool = True): preserve_order = pynutil.insert(" preserve_order: true") integer_plus_maj = graph_integer + insert_space + pynutil.insert(curr_symbol) @ graph_maj_plural - integer_plus_maj |= graph_integer_one + insert_space + pynutil.insert(curr_symbol) @ graph_maj_singular - # non zero integer part - integer_plus_maj = (pynini.closure(NEMO_DIGIT) - "0") @ integer_plus_maj + integer_plus_maj_with_one = integer_plus_maj | ( + graph_integer_one + insert_space + pynutil.insert(curr_symbol) @ graph_maj_singular + ) + # Amount == 1 without fractional part uses graph_integer_one_unit / graph_one_prefix. + integer_plus_maj_no_minor = (pynini.closure(NEMO_DIGIT) - "0") @ integer_plus_maj + integer_plus_maj_with_minor = (pynini.closure(NEMO_DIGIT) - "0") @ integer_plus_maj_with_one graph_fractional_one = two_digits_fractional_part @ pynini.cross("1", "") graph_fractional_one = pynutil.insert("fractional_part: \"") + graph_fractional_one + pynutil.insert("\"") @@ -141,22 +144,16 @@ def __init__(self, cardinal: GraphFst, deterministic: bool = True): graph_fractional_up_to_ten + insert_space + pynutil.insert(curr_symbol) @ graph_min_plural ) - graph_with_no_minor_curr = integer_plus_maj - graph_with_no_minor_curr |= pynutil.add_weight( - integer_plus_maj, - weight=0.0001, - ) - - graph_with_no_minor_curr = pynutil.delete(curr_symbol) + graph_with_no_minor_curr + preserve_order + graph_with_no_minor_curr = pynutil.delete(curr_symbol) + integer_plus_maj_no_minor + preserve_order graph_with_no_minor = ( graph_with_no_minor_curr if graph_with_no_minor is None else pynini.union(graph_with_no_minor, graph_with_no_minor_curr) ) - decimal_graph_with_minor_curr = integer_plus_maj + pynini.cross(".", " ") + fractional_plus_min + decimal_graph_with_minor_curr = integer_plus_maj_with_minor + pynini.cross(".", " ") + fractional_plus_min decimal_graph_with_minor_curr |= pynutil.add_weight( - integer_plus_maj + integer_plus_maj_with_minor + pynini.cross(".", " ") + pynutil.insert("fractional_part: \"") + two_digits_fractional_part @ cardinal_graph diff --git a/nemo_text_processing/text_normalization/ar/verbalizers/money.py b/nemo_text_processing/text_normalization/ar/verbalizers/money.py index 46da10742..9f5041b13 100644 --- a/nemo_text_processing/text_normalization/ar/verbalizers/money.py +++ b/nemo_text_processing/text_normalization/ar/verbalizers/money.py @@ -28,6 +28,7 @@ class MoneyFst(GraphFst): Finite state transducer for verbalizing money, e.g. money { integer_part: "تسعة" currency_maj: "يورو" preserve_order: true} -> "تسعة يورو" money { integer_part: "تسعة" currency_maj: "دولار" preserve_order: true} -> "تسعة دولار" + money { currency_maj: "دولار" integer_part: "واحد" morphosyntactic_features: "currency_first"} -> "دولار واحد" money { integer_part: "خمسة" currency_maj: "دينار كويتي"} -> "خمسة دينار كويتي" Args: @@ -49,9 +50,10 @@ def __init__(self, deterministic: bool = True): integer_part = pynutil.delete("integer_part: \"") + pynini.closure(NEMO_NOT_QUOTE, 1) + pynutil.delete("\"") add_and = pynutil.insert(" و") + morph_currency_first = pynutil.delete(' morphosyntactic_features: "currency_first"') - # *** currency_maj - graph_integer = maj + keep_space + integer_part + # currency_maj before integer_part; disambiguated via morphosyntactic_features for Sparrowhawk. + graph_currency_first = maj + keep_space + integer_part + delete_space + morph_currency_first # *** currency_maj + (***) (و) *** current_min graph_integer_with_minor = ( @@ -65,12 +67,10 @@ def __init__(self, deterministic: bool = True): + pynini.closure(keep_space + min, 0, 1) + delete_preserve_order ) - # this graph fix word order from dollar three (دولار تسعة)--> three dollar (تسعة دولار) graph_integer_no_minor = integer_part + keep_space + maj + delete_space + delete_preserve_order - # *** current_min graph_minor = fractional_part + keep_space + delete_space + min + delete_preserve_order - graph = graph_integer | graph_integer_with_minor | graph_minor | graph_integer_no_minor + graph = graph_currency_first | graph_integer_with_minor | graph_minor | graph_integer_no_minor delete_tokens = self.delete_tokens(graph) self.fst = delete_tokens.optimize() diff --git a/nemo_text_processing/text_normalization/en/taggers/serial.py b/nemo_text_processing/text_normalization/en/taggers/serial.py index f650c8ff3..4a5d6bb9d 100644 --- a/nemo_text_processing/text_normalization/en/taggers/serial.py +++ b/nemo_text_processing/text_normalization/en/taggers/serial.py @@ -18,6 +18,8 @@ from pynini.lib import pynutil from nemo_text_processing.text_normalization.en.graph_utils import ( + MIN_NEG_WEIGHT, + MIN_POS_WEIGHT, NEMO_ALPHA, NEMO_DIGIT, NEMO_NOT_SPACE, @@ -28,16 +30,65 @@ from nemo_text_processing.text_normalization.en.utils import get_abs_path, load_labels +def _leading_zero_graph(cardinal: GraphFst) -> "pynini.FstLike": + return pynini.compose(pynini.accep("0") + pynini.closure(NEMO_DIGIT), cardinal.single_digits_graph).optimize() + + +def _build_serial_graph( + num_graph: "pynini.FstLike", + delimiter: "pynini.FstLike", + alphas: "pynini.FstLike", + ordinal: GraphFst, +) -> "pynini.FstLike": + letter_num = alphas + delimiter + num_graph + num_letter = pynini.closure(num_graph + delimiter, 1) + alphas + next_alpha_or_num = pynini.closure(delimiter + (alphas | num_graph)) + next_alpha_or_num |= pynini.closure( + delimiter + + num_graph + + plurals._priority_union(pynini.accep(" "), pynutil.insert(" "), NEMO_SIGMA).optimize() + + alphas + ) + + serial_graph = letter_num + next_alpha_or_num + serial_graph |= num_letter + next_alpha_or_num + serial_graph |= num_graph + delimiter + num_graph + delimiter + num_graph + pynini.closure(delimiter + num_graph) + + symbols = [x[0] for x in load_labels(get_abs_path("data/whitelist/symbol.tsv"))] + symbols = pynini.union(*symbols) + serial_graph |= pynini.compose(NEMO_SIGMA + symbols + NEMO_SIGMA, num_graph + delimiter + num_graph) + + serial_graph = pynini.compose( + pynini.difference(NEMO_SIGMA, pynini.project(ordinal.graph, "input")), serial_graph + ).optimize() + + serial_graph = pynutil.add_weight(serial_graph, MIN_POS_WEIGHT) + serial_graph |= ( + pynini.closure(NEMO_NOT_SPACE, 1) + (pynini.cross("^2", " squared") | pynini.cross("^3", " cubed")).optimize() + ) + + serial_graph = ( + pynini.closure((serial_graph | num_graph | alphas) + delimiter) + + serial_graph + + pynini.closure(delimiter + (serial_graph | num_graph | alphas)) + ) + return serial_graph.optimize() + + class SerialFst(GraphFst): """ - This class is a composite class of two other class instances + Finite state transducer for classifying serial numbers without conventional delimiters. + + Digit normalization within letter-digit tokens follows: + 1. 1-2 digits, or single digits followed by zeros -> cardinal + 2. 3 digits not ending in 00, or 4+ digits -> single-digit reading + 3. Digit-only tokens separated by ``/`` -> cardinal per segment (5+ digits stay single-digit) Args: - time: composed tagger and verbalizer - date: composed tagger and verbalizer - cardinal: tagger + cardinal: cardinal tagger + ordinal: ordinal tagger (used to exclude ordinal readings) deterministic: if True will provide a single transduction option, - for False multiple transduction are generated (used for audio-based normalization) + for False multiple transduction are generated (used for audio-based normalization) lm: whether to use for hybrid LM """ @@ -48,31 +99,56 @@ def __init__(self, cardinal: GraphFst, ordinal: GraphFst, deterministic: bool = Finite state transducer for classifying serial (handles only cases without delimiters, values with delimiters are handled by default). The serial is a combination of digits, letters and dashes, e.g.: - c325b -> tokens { cardinal { integer: "c three two five b" } } + "H800" -> tokens { name: "H eight hundred" } + "a320b" -> tokens { name: "a three two zero b" } + "12/345/67890" -> tokens { name: "twelve/three hundred forty five/six seven eight nine zero" } + """ if deterministic: - num_graph = pynini.compose(NEMO_DIGIT ** (6, ...), cardinal.single_digits_graph).optimize() - num_graph |= pynini.compose(NEMO_DIGIT ** (1, 5), cardinal.graph).optimize() - # to handle numbers starting with zero - num_graph |= pynini.compose( - pynini.accep("0") + pynini.closure(NEMO_DIGIT), cardinal.single_digits_graph + num_graph_pure = ( + pynini.compose(NEMO_DIGIT ** (1, 3), cardinal.graph) + | pynini.compose(NEMO_DIGIT ** (4, ...), cardinal.single_digits_graph) + | _leading_zero_graph(cardinal) + ).optimize() + + num_graph_alnum = ( + pynini.compose(NEMO_DIGIT, cardinal.graph) + | pynini.compose(NEMO_DIGIT**2, cardinal.graph) + | pynutil.add_weight( + pynini.compose(NEMO_DIGIT + pynini.closure("0", 1), cardinal.graph), MIN_NEG_WEIGHT + ) + | pynini.compose( + pynini.difference(NEMO_DIGIT**3, NEMO_DIGIT + NEMO_DIGIT + "00"), cardinal.single_digits_graph + ) + | pynini.compose(NEMO_DIGIT ** (4, ...), cardinal.single_digits_graph) + | _leading_zero_graph(cardinal) + ).optimize() + + num_graph_slash = ( + pynini.compose(NEMO_DIGIT ** (1, 4), cardinal.graph) + | pynini.compose(NEMO_DIGIT ** (5, ...), cardinal.single_digits_graph) + | _leading_zero_graph(cardinal) ).optimize() + else: - num_graph = cardinal.final_graph + num_graph_pure = cardinal.final_graph + num_graph_alnum = cardinal.final_graph + num_graph_slash = cardinal.final_graph # TODO: "#" doesn't work from the file symbols_graph = pynini.string_file(get_abs_path("data/whitelist/symbol.tsv")).optimize() | pynini.cross( "#", "hash" ) - num_graph |= symbols_graph + num_graph_pure |= symbols_graph + num_graph_alnum |= symbols_graph if not self.deterministic and not lm: - num_graph |= cardinal.single_digits_graph - num_graph |= pynini.compose(num_graph, NEMO_SIGMA + pynutil.delete("hundred ") + NEMO_SIGMA) - # also allow double digits to be pronounced as integer in serial number - num_graph |= pynutil.add_weight( - NEMO_DIGIT**2 @ cardinal.graph_hundred_component_at_least_one_none_zero_digit, weight=0.0001 + num_graph_pure |= cardinal.single_digits_graph + num_graph_pure |= pynini.compose(num_graph_pure, NEMO_SIGMA + pynutil.delete("hundred ") + NEMO_SIGMA) + num_graph_pure |= pynutil.add_weight( + NEMO_DIGIT**2 @ cardinal.graph_hundred_component_at_least_one_none_zero_digit, weight=MIN_POS_WEIGHT ) + num_graph_alnum = num_graph_pure # add space between letter and digit/symbol symbols = [x[0] for x in load_labels(get_abs_path("data/whitelist/symbol.tsv"))] @@ -90,44 +166,21 @@ def __init__(self, cardinal: GraphFst, ordinal: GraphFst, deterministic: bool = delimiter |= pynini.cross("-", " dash ") | pynini.cross("/", " slash ") alphas = pynini.closure(NEMO_ALPHA, 1) - letter_num = alphas + delimiter + num_graph - num_letter = pynini.closure(num_graph + delimiter, 1) + alphas - next_alpha_or_num = pynini.closure(delimiter + (alphas | num_graph)) - next_alpha_or_num |= pynini.closure( - delimiter - + num_graph - + plurals._priority_union(pynini.accep(" "), pynutil.insert(" "), NEMO_SIGMA).optimize() - + alphas - ) - - serial_graph = letter_num + next_alpha_or_num - serial_graph |= num_letter + next_alpha_or_num - # numbers only with 2+ delimiters - serial_graph |= ( - num_graph + delimiter + num_graph + delimiter + num_graph + pynini.closure(delimiter + num_graph) - ) - # 2+ symbols - serial_graph |= pynini.compose(NEMO_SIGMA + symbols + NEMO_SIGMA, num_graph + delimiter + num_graph) - - # exclude ordinal numbers from serial options - serial_graph = pynini.compose( - pynini.difference(NEMO_SIGMA, pynini.project(ordinal.graph, "input")), serial_graph - ).optimize() - serial_graph = pynutil.add_weight(serial_graph, 0.0001) - serial_graph |= ( - pynini.closure(NEMO_NOT_SPACE, 1) - + (pynini.cross("^2", " squared") | pynini.cross("^3", " cubed")).optimize() - ) + serial_graph = _build_serial_graph(num_graph_pure, delimiter, alphas, ordinal) + serial_graph_alnum = _build_serial_graph(num_graph_alnum, delimiter, alphas, ordinal) - # at least one serial graph with alpha numeric value and optional additional serial/num/alpha values - serial_graph = ( - pynini.closure((serial_graph | num_graph | alphas) + delimiter) - + serial_graph - + pynini.closure(delimiter + (serial_graph | num_graph | alphas)) + # Rule 3: tokens that contain only digits and slashes (e.g. 31/31/100, 123/261788/2021). + slash_digit_token = ( + pynini.closure(NEMO_DIGIT, 1) + pynini.accep("/") + pynini.closure(NEMO_DIGIT | pynini.accep("/"), 0) ) + slash_serial = pynini.compose( + slash_digit_token, + pynini.closure(num_graph_slash + pynini.accep("/"), 1) + num_graph_slash, + ).optimize() + serial_graph |= pynutil.add_weight(slash_serial, MIN_NEG_WEIGHT) - serial_graph |= pynini.compose(graph_with_space, serial_graph.optimize()).optimize() + serial_graph |= pynini.compose(graph_with_space, serial_graph_alnum.optimize()).optimize() serial_graph = pynini.compose(pynini.closure(NEMO_NOT_SPACE, 2), serial_graph).optimize() # this is not to verbolize "/" as "slash" in cases like "import/export" diff --git a/nemo_text_processing/text_normalization/hi/data/address/context.tsv b/nemo_text_processing/text_normalization/hi/data/address/context.tsv index 9faadaa3b..d57bfd7d3 100644 --- a/nemo_text_processing/text_normalization/hi/data/address/context.tsv +++ b/nemo_text_processing/text_normalization/hi/data/address/context.tsv @@ -44,5 +44,4 @@ वेस्ट सामने पीछे -वीया -आर डी \ No newline at end of file +वीया \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/address/en_to_hi_mapping.tsv b/nemo_text_processing/text_normalization/hi/data/address/en_to_hi_mapping.tsv deleted file mode 100644 index 15929b547..000000000 --- a/nemo_text_processing/text_normalization/hi/data/address/en_to_hi_mapping.tsv +++ /dev/null @@ -1,2 +0,0 @@ -street स्ट्रीट -southern सदर्न \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/date/days.tsv b/nemo_text_processing/text_normalization/hi/data/date/days.tsv index 633e2aec0..6df0fa3d4 100644 --- a/nemo_text_processing/text_normalization/hi/data/date/days.tsv +++ b/nemo_text_processing/text_normalization/hi/data/date/days.tsv @@ -1,40 +1,9 @@ -०१ एक -०२ दो -०३ तीन -०४ चार -०५ पाँच -०६ छः -०७ सात -०८ आठ -०९ नौ -१० दस -११ ग्यारह -१२ बारह -१३ तेरह -१४ चौदह -१५ पंद्रह -१६ सोलह -१७ सत्रह -१८ अठारह -१९ उन्नीस -२० बीस -२१ इक्कीस -२२ बाईस -२३ तेईस -२४ चौबीस -२५ पच्चीस -२६ छब्बीस -२७ सत्ताईस -२८ अट्ठाईस -२९ उनतीस -३० तीस -३१ इकतीस 01 एक 02 दो 03 तीन 04 चार 05 पाँच -06 छः +06 छह 07 सात 08 आठ 09 नौ @@ -59,4 +28,35 @@ 28 अट्ठाईस 29 उनतीस 30 तीस -31 इकतीस \ No newline at end of file +31 इकतीस +०१ एक +०२ दो +०३ तीन +०४ चार +०५ पाँच +०६ छह +०७ सात +०८ आठ +०९ नौ +१० दस +११ ग्यारह +१२ बारह +१३ तेरह +१४ चौदह +१५ पंद्रह +१६ सोलह +१७ सत्रह +१८ अठारह +१९ उन्नीस +२० बीस +२१ इक्कीस +२२ बाईस +२३ तेईस +२४ चौबीस +२५ पच्चीस +२६ छब्बीस +२७ सत्ताईस +२८ अट्ठाईस +२९ उनतीस +३० तीस +३१ इकतीस \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/date/months.tsv b/nemo_text_processing/text_normalization/hi/data/date/months.tsv index af770dafc..3667f07cf 100644 --- a/nemo_text_processing/text_normalization/hi/data/date/months.tsv +++ b/nemo_text_processing/text_normalization/hi/data/date/months.tsv @@ -1,17 +1,5 @@ -०१ जनवरी -०२ फ़रवरी -०३ मार्च -०४ अप्रैल -०५ मई -०६ जून -०७ जुलाई -०८ अगस्त -०९ सितंबर -१० अक्टूबर -११ नवंबर -१२ दिसंबर 01 जनवरी -02 फ़रवरी +02 फरवरी 03 मार्च 04 अप्रैल 05 मई @@ -21,4 +9,16 @@ 09 सितंबर 10 अक्टूबर 11 नवंबर -12 दिसंबर \ No newline at end of file +12 दिसंबर +०१ जनवरी +०२ फरवरी +०३ मार्च +०४ अप्रैल +०५ मई +०६ जून +०७ जुलाई +०८ अगस्त +०९ सितंबर +१० अक्टूबर +११ नवंबर +१२ दिसंबर \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/date/prefixes.tsv b/nemo_text_processing/text_normalization/hi/data/date/prefixes.tsv index d4c1ca0b1..6166ec327 100644 --- a/nemo_text_processing/text_normalization/hi/data/date/prefixes.tsv +++ b/nemo_text_processing/text_normalization/hi/data/date/prefixes.tsv @@ -1,3 +1,4 @@ -सन् -सन -साल \ No newline at end of file +सन् +सन +साल +दशक \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/electronic/common_words.tsv b/nemo_text_processing/text_normalization/hi/data/electronic/common_words.tsv deleted file mode 100644 index ef3cf696f..000000000 --- a/nemo_text_processing/text_normalization/hi/data/electronic/common_words.tsv +++ /dev/null @@ -1,164 +0,0 @@ -about अबाउट -blog ब्लॉग -home होम -index इंडेक्स -login लॉगिन -register रजिस्टर -search सर्च -tags टैग्स -category केटेगरी -categories केटेगरीज़ -post पोस्ट -posts पोस्ट्स -page पेज -pages पेजेस -user यूज़र -users यूज़र्स -Users यूज़र्स -User यूज़र -Desktop डेस्कटॉप -Documents डॉक्युमेंट्स -Downloads डाउनलोड्स -Music म्यूज़िक -Pictures पिक्चर्स -Videos वीडियोज़ -admin एडमिन -app ऐप -faq एफ ए क्यू -help हेल्प -terms टर्म्स -privacy प्राइवेसी -contact कॉन्टैक्ट -main मेन -explore एक्सप्लोर -wiki विकी -docs डॉक्स -download डाउनलोड -downloads डाउनलोड्स -upload अपलोड -uploads अपलोड्स -photos फ़ोटोज़ -images इमेजेज़ -music म्यूज़िक -video वीडियो -videos वीडियोज़ -desktop डेस्कटॉप -documents डॉक्युमेंट्स -master मास्टर -blob ब्लॉब -tree ट्री -tests टेस्ट्स -test टेस्ट -config कॉन्फ़िग -settings सेटिंग्स -profile प्रोफ़ाइल -account अकाउंट -web वेब -email ई मेल -laptop लैपटॉप -mobile मोबाइल -phone फोन -phones फोन्स -online ऑनलाइन -courses कोर्सेज़ -learn लर्न -learning लर्निंग -university यूनिवर्सिटी -academy अकेडमी -domain डोमेन -domains डोमेन्स -analysis अनैलिसिस -play प्ले -maps मैप्स -drive ड्राइव -cloud क्लाउड -services सर्विसेज़ -india इंडिया -screenshot स्क्रीनशॉट -zoom ज़ूम -audacity ऑडेसिटी -coursera कोर्सेरा -apache अपाची -kernel कर्नल -bin बिन -var वार -home होम -activate एक्टिवेट -sites साइट्स -available अवेलेबल -enabled इनेबल्ड -backups बैकअप्स -temp टेम्प -secure सेक्योर -real रियल -data डेटा -work वर्क -survey सर्वे -files फाइल्स -file फ़ाइल -chapter चैप्टर -python पाइथन -audio ऑडियो -sample सैंपल -templates टेम्पलेट्स -impress इंप्रेस -office ऑफिस -libreoffice लिब्रे ऑफिस -express एक्सप्रेस -scribe स्क्राइब -transcription ट्रांसक्रिप्शन -software सॉफ्टवेयर -teams टीम्स -school स्कूल -space स्पेस -green ग्रीन -brown ब्राउन -white व्हाइट -black ब्लैक -homepage होमपेज -content कंटेन्ट -default डिफ़ॉल्ट -foodhealth फूड हेल्थ -workflows वर्कफ्लोज़ -world वर्ल्ड -list लिस्ट -and एंड -LICENSE लाइसेंस -license लाइसेंस -tag टैग -blogs ब्लॉग्स -bath बाथ -ward वार्ड -banks बैंक्स -Audio ऑडियो -dean डीन -rice राइस -honda होंडा -ford फोर्ड -house हाउस -bharat भरत -rich रिच -cook कुक -lane लेन -knight नाइट -moody मूडी -wise वाइज़ -shields शील्ड्स -puppy पप्पी -recipe रेसिपी -hall हॉल -mason मेसन -king किंग -fry फ्राई -flowers फ्लावर्स -assam आसाम -grace ग्रेस -bishop बिशप -woods वुड्स -brewer ब्रूअर -cannon कैनन -saute सौटे -pope पोप -robin रॉबिन -price प्राइस -address एड्रेस \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/electronic/domain.tsv b/nemo_text_processing/text_normalization/hi/data/electronic/domain.tsv index 2b4f70f03..dccb5dd90 100644 --- a/nemo_text_processing/text_normalization/hi/data/electronic/domain.tsv +++ b/nemo_text_processing/text_normalization/hi/data/electronic/domain.tsv @@ -1,38 +1,24 @@ -com कॉम -org ऑर्ग -net नेट -edu ई डी यू -gov जी ओ वी -biz बिज़ -info इन्फो -in इन -co सी ओ -io आई ओ -ai ए आई -uk यू के -us यू एस -ru आर यू -de डी ई -fr एफ आर -jp जे पी -cn सी एन -au ए यू -ca सी ए -br बी आर -ac ए सी -res आर ई एस -nic एन आई सी -ernet ई आर नेट -mil एम आई एल -int आई एन टी -tv टी वी -me एम ई -tech टेक -dev डी ई वी -app ऐप -xyz एक्स वाई ज़ेड -online ऑनलाइन -store स्टोर -blog ब्लॉग -site साइट -pro प्रो \ No newline at end of file +com +org +net +edu +gov +in +co +io +ai +uk +us +au +ca +ac +res +nic +ernet +tv +me +tech +dev +app +biz +info diff --git a/nemo_text_processing/text_normalization/hi/data/electronic/elements.tsv b/nemo_text_processing/text_normalization/hi/data/electronic/elements.tsv new file mode 100644 index 000000000..be4610634 --- /dev/null +++ b/nemo_text_processing/text_normalization/hi/data/electronic/elements.tsv @@ -0,0 +1,132 @@ +Ac +Ag +Al +Am +An +Ar +As +At +Au +B +Ba +Be +Bh +Bi +Bk +Br +Bu +Bz +C +Ca +Cd +Ce +Cf +Cl +Cm +Cn +Co +Cp +Cr +Cs +Cu +D +Db +Ds +Dy +En +Er +Es +Et +Eu +F +Fe +Fl +Fm +Fr +Ga +Gd +Ge +H +He +Hf +Hg +Ho +Hs +I +In +Ir +K +Kr +La +Li +Ln +Lr +Lu +Lv +M +Mc +Md +Me +Mg +Mn +Mo +Mt +N +Na +Nb +Nd +Ne +Nh +Ni +No +Np +O +Og +Os +P +Pa +Pb +Pd +Ph +Pm +Po +Pr +Pt +Pu +R +Ra +Rb +Re +Rf +Rg +Rh +Rn +Ru +S +Sb +Sc +Se +Sg +Si +Sm +Sn +Sr +T +Ta +Tb +Tc +Te +Th +Ti +Tl +Tm +Ts +U +V +W +X +Xe +Y +Yb +Zn +Zr \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/electronic/file_extensions.tsv b/nemo_text_processing/text_normalization/hi/data/electronic/file_extensions.tsv index 2c261a062..7283febf5 100644 --- a/nemo_text_processing/text_normalization/hi/data/electronic/file_extensions.tsv +++ b/nemo_text_processing/text_normalization/hi/data/electronic/file_extensions.tsv @@ -1,51 +1,34 @@ -jpg जे पी जी -jpeg जे पी ई जी -png पी एन जी -gif जी आई एफ -pdf पी डी एफ -doc डी ओ सी -docx डी ओ सी एक्स -xls एक्स एल एस -xlsx एक्स एल एस एक्स -ppt पी पी टी -pptx पी पी टी एक्स -csv सी एस वी -txt टी एक्स टी -html एच टी एम एल -htm एच टी एम -xml एक्स एम एल -json जे एस ओ एन -css सी एस एस -js जे एस -py पी वाई -java जावा -cpp सी पी पी -zip ज़िप -rar आर ए आर -tar टी ए आर -mp3 एम पी तीन -mp4 एम पी चार -avi ए वी आई -mkv एम के वी -mov एम ओ वी -wav डब्ल्यू ए वी -svg एस वी जी -ico आई सी ओ -apk ए पी के -exe ई एक्स ई -dmg डी एम जी -iso आई एस ओ -sql एस क्यू एल -log एल ओ जी -bak बी ए के -JPG जे पी जी -JPEG जे पी ई जी -PNG पी एन जी -PDF पी डी एफ -DOC डी ओ सी -DOCX डी ओ सी एक्स -CSV सी एस वी -TXT टी एक्स टी -HTML एच टी एम एल -MP3 एम पी तीन -MP4 एम पी चार \ No newline at end of file +jpg +jpeg +png +gif +pdf +doc +docx +xls +xlsx +ppt +pptx +csv +txt +html +xml +json +css +js +py +java +cpp +zip +rar +tar +mp3 +mp4 +avi +mkv +mov +wav +svg +apk +exe +sql \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/electronic/protocols.tsv b/nemo_text_processing/text_normalization/hi/data/electronic/protocols.tsv index 627781003..dd897a4b6 100644 --- a/nemo_text_processing/text_normalization/hi/data/electronic/protocols.tsv +++ b/nemo_text_processing/text_normalization/hi/data/electronic/protocols.tsv @@ -1,5 +1,5 @@ -https एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश -http एच टी टी पी कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश -www डब्ल्यू डब्ल्यू डब्ल्यू डॉट -httpswww एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश डब्ल्यू डब्ल्यू डब्ल्यू डॉट -httpwww एच टी टी पी कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश डब्ल्यू डब्ल्यू डब्ल्यू डॉट \ No newline at end of file +https https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश +http http कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश +www www डॉट +httpswww https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश www डॉट +httpwww http कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश www डॉट diff --git a/nemo_text_processing/text_normalization/hi/data/electronic/server_name.tsv b/nemo_text_processing/text_normalization/hi/data/electronic/server_name.tsv deleted file mode 100644 index 8d986bfd8..000000000 --- a/nemo_text_processing/text_normalization/hi/data/electronic/server_name.tsv +++ /dev/null @@ -1,64 +0,0 @@ -gmail जीमेल -yahoo याहू -hotmail हॉटमेल -outlook आउटलुक -live लाइव -google गूगल -microsoft माइक्रोसॉफ्ट -facebook फ़ेसबुक -twitter ट्विटर -instagram इंस्टाग्राम -linkedin लिंक्डइन -youtube यूट्यूब -amazon अमेज़ोन -wikipedia विकिपीडिया -github गिटहब -reddit रेडिट -netflix नेटफ्लिक्स -spotify स्पॉटिफाई -apple एप्पल -samsung सैमसंग -nvidia एनविडिया -intel इंटेल -adobe अडोब -wordpress वर्डप्रेस -blogger ब्लॉगर -mentalfloss मेंटल फ्लॉस -placekitten प्लेस किटन -dummyimage डमी इमेज -reliablesoft रिलाएबल सॉफ्ट -ebay ई बे -moz मोज़ -mozilla मॉज़िला -genius जीनियस -groupon ग्रुप ऑन -gutenberg गुटेनबर्ग -recipepuppy रेसिपी पप्पी -buyagift बाय अ गिफ्ट -webmd वेब एम डी -researchgate रिसर्च गेट -afternic आफ्टर निक -hipolabs हिपो लैब्स -licindia एल आई सी इंडिया -placeimg प्लेस आई एम जी -codecademy कोड कैडेमी -skillshop स्किलशॉप -skillshare स्किल शेयर -udemy यूडेमी -masterclass मास्टरक्लास -amity एमिटी -sharda शारदा -universities यूनिवर्सिटीज़ -mcdonald मैक्डॉनल्ड -southmountaincc साउथ माउन्टेन सी सी -academyart अकेडमी आर्ट -bryanuniversity ब्रायन यूनिवर्सिटी -centralaz सेंट्रल ए ज़ेड -alaska अलास्का -phoenix फीनिक्स -phoenixcollege फीनिक्स कॉलेज -maricopa मैरीकोपा -prescott प्रेसकॉट -azwestern ए ज़ेड वेस्टर्न -poetry पोएट्री -harvard हावर्ड \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/electronic/symbol_classes.tsv b/nemo_text_processing/text_normalization/hi/data/electronic/symbol_classes.tsv new file mode 100644 index 000000000..cf17c8756 --- /dev/null +++ b/nemo_text_processing/text_normalization/hi/data/electronic/symbol_classes.tsv @@ -0,0 +1,16 @@ +. email,url,unix,windows +- email,url,unix,windows,chem +_ email,url,unix,windows +/ url,unix +$ unix +\ windows +( windows,chem +) windows,chem ++ url,chem +– chem +# url +? url +& url += url +% url +: url,windows \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/electronic/symbols.tsv b/nemo_text_processing/text_normalization/hi/data/electronic/symbols.tsv index fe8881ae8..e720f5338 100644 --- a/nemo_text_processing/text_normalization/hi/data/electronic/symbols.tsv +++ b/nemo_text_processing/text_normalization/hi/data/electronic/symbols.tsv @@ -30,4 +30,7 @@ $ डॉलर \[ ओपन स्क्वेर ब्रेकेट \] क्लोज़ स्क्वेर ब्रेकेट { ओपन कर्ली ब्रेकेट -} क्लोज़ कर्ली ब्रेकेट \ No newline at end of file +} क्लोज़ कर्ली ब्रेकेट +– माइनस +⁻ माइनस +⁺ प्लस \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/fraction/__init__.py b/nemo_text_processing/text_normalization/hi/data/fraction/__init__.py new file mode 100644 index 000000000..4fc25d0d3 --- /dev/null +++ b/nemo_text_processing/text_normalization/hi/data/fraction/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/text_normalization/hi/data/fraction/common_fractions.tsv b/nemo_text_processing/text_normalization/hi/data/fraction/common_fractions.tsv new file mode 100644 index 000000000..5e44cb502 --- /dev/null +++ b/nemo_text_processing/text_normalization/hi/data/fraction/common_fractions.tsv @@ -0,0 +1,10 @@ +१/२ आधा +१/३ तिहाई +२/३ दो तिहाई +१/४ चौथाई +३/४ तीन चौथाई +1/2 आधा +1/3 तिहाई +2/3 दो तिहाई +1/4 चौथाई +3/4 तीन चौथाई \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/measure/unit.tsv b/nemo_text_processing/text_normalization/hi/data/measure/unit.tsv index 4065bc86b..d236dd51b 100644 --- a/nemo_text_processing/text_normalization/hi/data/measure/unit.tsv +++ b/nemo_text_processing/text_normalization/hi/data/measure/unit.tsv @@ -1,6 +1,5 @@ °C डिग्री सेल्सियस °F डिग्री फारेनहाइट -K केल्विन g ग्राम kg किलोग्राम mg मिलीग्राम @@ -72,7 +71,6 @@ dl² वर्ग डेसीलीटर dal² वर्ग डेकालीटर dl³ घन डेसीलीटर dal³ घन डेकालीटर -L लीटर kL किलोलीटर mL मिलीलीटर mL² वर्ग मिलीलीटर @@ -99,12 +97,10 @@ mm³ घन मिलीमीटर qt क्वार्ट gal गैलन pt पिंट -W वाट MW मेगावाट KW किलोवाट b बिट Mb मेगाबिट -B बाइट GB गीगाबाइट KB किलोबाइट TB टेराबाइट @@ -114,11 +110,7 @@ EB एक्साबाइट ZB जेटाबाइट YB योटाबाइट BB ब्रोन्टोबाइट -C कूलंब -V वोल्ट Pa पास्कल -A ऐंपीयर -J जूल s सेकंड hr घंटा h घंटे @@ -131,7 +123,6 @@ doz दर्जन Hz हर्ट्ज़ GHz गीगाहर्ट्ज़ KHz किलोहर्ट्ज़ -N न्यूटन dB डेसीबल yr साल hp हॉर्सपॉवर @@ -151,5 +142,4 @@ mi/hr मील प्रति घंटा mi/min मील प्रति मिनट ₹/ac रुपए प्रति एकड़ x बाई -X बाई -* बाई +* बाई \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/money/currency_singular.tsv b/nemo_text_processing/text_normalization/hi/data/money/currency_singular.tsv new file mode 100644 index 000000000..af8d793f2 --- /dev/null +++ b/nemo_text_processing/text_normalization/hi/data/money/currency_singular.tsv @@ -0,0 +1,9 @@ +₹ रुपया +£ पाउंड +₩ वॉन +$ डॉलर +₺ लीरा +৳ टका +¥ येन +₦ नाइरा +€ यूरो \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/money/major_minor_currencies.tsv b/nemo_text_processing/text_normalization/hi/data/money/major_minor_currencies.tsv index cf62891d1..a9186acc3 100644 --- a/nemo_text_processing/text_normalization/hi/data/money/major_minor_currencies.tsv +++ b/nemo_text_processing/text_normalization/hi/data/money/major_minor_currencies.tsv @@ -1,4 +1,5 @@ रुपए पैसे +रुपया पैसे पाउंड पेंस वॉन जिओन डॉलर सेंट @@ -6,4 +7,4 @@ टका पैसे येन सेन नाइरा कोबो -यूरो सेंट +यूरो सेंट \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/ordinal/suffixes_map.tsv b/nemo_text_processing/text_normalization/hi/data/ordinal/suffixes_map.tsv index 77139cff5..2abb5c492 100644 --- a/nemo_text_processing/text_normalization/hi/data/ordinal/suffixes_map.tsv +++ b/nemo_text_processing/text_normalization/hi/data/ordinal/suffixes_map.tsv @@ -1,2 +1 @@ -वे वें - +वे वें \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/roman/__init__.py b/nemo_text_processing/text_normalization/hi/data/roman/__init__.py new file mode 100644 index 000000000..4fc25d0d3 --- /dev/null +++ b/nemo_text_processing/text_normalization/hi/data/roman/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/text_normalization/hi/data/roman/roman_ordinal_exceptions.tsv b/nemo_text_processing/text_normalization/hi/data/roman/roman_ordinal_exceptions.tsv new file mode 100644 index 000000000..b298e13e6 --- /dev/null +++ b/nemo_text_processing/text_normalization/hi/data/roman/roman_ordinal_exceptions.tsv @@ -0,0 +1,10 @@ +Iला पहला +Iली पहली +IIरा दूसरा +IIरी दूसरी +IIIरा तीसरा +IIIरी तीसरी +IVथा चौथा +IVथी चौथी +VIठा छठा +VIठी छठी \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/roman/roman_to_spoken.tsv b/nemo_text_processing/text_normalization/hi/data/roman/roman_to_spoken.tsv new file mode 100644 index 000000000..69b760196 --- /dev/null +++ b/nemo_text_processing/text_normalization/hi/data/roman/roman_to_spoken.tsv @@ -0,0 +1,100 @@ +I एक +II दो +III तीन +IV चार +V पाँच +VI छह +VII सात +VIII आठ +IX नौ +X दस +XI ग्यारह +XII बारह +XIII तेरह +XIV चौदह +XV पंद्रह +XVI सोलह +XVII सत्रह +XVIII अठारह +XIX उन्नीस +XX बीस +XXI इक्कीस +XXII बाईस +XXIII तेईस +XXIV चौबीस +XXV पच्चीस +XXVI छब्बीस +XXVII सत्ताईस +XXVIII अट्ठाईस +XXIX उनतीस +XXX तीस +XXXI इकतीस +XXXII बत्तीस +XXXIII तैंतीस +XXXIV चौंतीस +XXXV पैंतीस +XXXVI छत्तीस +XXXVII सैंतीस +XXXVIII अड़तीस +XXXIX उनचालीस +XL चालीस +XLI इकतालीस +XLII बयालीस +XLIII तैंतालीस +XLIV चौंतालीस +XLV पैंतालीस +XLVI छियालीस +XLVII सैंतालीस +XLVIII अड़तालीस +XLIX उनचास +L पचास +LI इक्यावन +LII बावन +LIII तिरपन +LIV चौवन +LV पचपन +LVI छप्पन +LVII सत्तावन +LVIII अट्ठावन +LIX उनसठ +LX साठ +LXI इकसठ +LXII बासठ +LXIII तिरसठ +LXIV चौंसठ +LXV पैंसठ +LXVI छियासठ +LXVII सड़सठ +LXVIII अड़सठ +LXIX उनहत्तर +LXX सत्तर +LXXI इकहत्तर +LXXII बहत्तर +LXXIII तिहत्तर +LXXIV चौहत्तर +LXXV पचहत्तर +LXXVI छिहत्तर +LXXVII सतहत्तर +LXXVIII अठहत्तर +LXXIX उनासी +LXXX अस्सी +LXXXI इक्यासी +LXXXII बयासी +LXXXIII तिरासी +LXXXIV चौरासी +LXXXV पचासी +LXXXVI छियासी +LXXXVII सत्तासी +LXXXVIII अट्ठासी +LXXXIX नवासी +XC नब्बे +XCI इक्यानवे +XCII बानवे +XCIII तिरानवे +XCIV चौरानवे +XCV पचानवे +XCVI छियानवे +XCVII सत्तानवे +XCVIII अट्ठानवे +XCIX निन्यानवे +C एक सौ \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/serial/__init__.py b/nemo_text_processing/text_normalization/hi/data/serial/__init__.py new file mode 100644 index 000000000..4fc25d0d3 --- /dev/null +++ b/nemo_text_processing/text_normalization/hi/data/serial/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/text_normalization/hi/data/serial/chars.tsv b/nemo_text_processing/text_normalization/hi/data/serial/chars.tsv new file mode 100644 index 000000000..d7c9c39e8 --- /dev/null +++ b/nemo_text_processing/text_normalization/hi/data/serial/chars.tsv @@ -0,0 +1,68 @@ +अ +आ +इ +ई +उ +ऊ +ऋ +ए +ऐ +ओ +औ +ऑ +ा +ि +ी +ु +ू +ृ +े +ै +ो +ौ +ॉ +ं +ः +ँ +क +ख +ग +घ +ङ +च +छ +ज +झ +ञ +ट +ठ +ड +ढ +ण +त +थ +द +ध +न +प +फ +ब +भ +म +य +र +ल +व +श +ष +स +ह +क़ +ख़ +ग़ +ज़ +ड़ +ढ़ +फ़ +य़ +् \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/serial/power_special.tsv b/nemo_text_processing/text_normalization/hi/data/serial/power_special.tsv new file mode 100644 index 000000000..64583f947 --- /dev/null +++ b/nemo_text_processing/text_normalization/hi/data/serial/power_special.tsv @@ -0,0 +1,4 @@ +^2 स्क्वेर्ड +^२ स्क्वेर्ड +^3 क्यूब +^३ क्यूब \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/serial/special_symbols.tsv b/nemo_text_processing/text_normalization/hi/data/serial/special_symbols.tsv new file mode 100644 index 000000000..c96a15bd6 --- /dev/null +++ b/nemo_text_processing/text_normalization/hi/data/serial/special_symbols.tsv @@ -0,0 +1,4 @@ +# हैशटैग +% प्रतिशत +& एंड +@ एट \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/taggers/cardinal.py b/nemo_text_processing/text_normalization/hi/taggers/cardinal.py index c29ccaa59..dd4611010 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/cardinal.py +++ b/nemo_text_processing/text_normalization/hi/taggers/cardinal.py @@ -18,6 +18,7 @@ from nemo_text_processing.text_normalization.hi.graph_utils import ( NEMO_ALL_DIGIT, NEMO_ALL_ZERO, + NEMO_DIGIT, GraphFst, insert_space, ) @@ -348,11 +349,46 @@ def create_larger_number_graph(digit_graph, suffix, zeros_counts, sub_graph): ) cardinal_with_leading_zeros = pynutil.add_weight(cardinal_with_leading_zeros, 0.5) + # Handle large numbers written with digit-group separators. + delete_separator = pynutil.delete(",") + two_digits = NEMO_ALL_DIGIT + NEMO_ALL_DIGIT + three_digits = NEMO_ALL_DIGIT + NEMO_ALL_DIGIT + NEMO_ALL_DIGIT + # Indian grouping: 1-2 leading digits, groups of 2, final group of 3. + indian_grouping = ( + pynini.closure(NEMO_ALL_DIGIT, 1, 2) + + pynini.closure(delete_separator + two_digits) + + delete_separator + + three_digits + ) + # International grouping: 1-3 leading digits, one or more groups of 3. + western_grouping = pynini.closure(NEMO_ALL_DIGIT, 1, 3) + pynini.closure(delete_separator + three_digits, 1) + strip_separators = (indian_grouping | western_grouping).optimize() + cardinal_with_separators = pynini.compose(strip_separators, graph_without_leading_zeros).optimize() + # Full graph including leading zeros - for standalone cardinal matching - final_graph = graph_without_leading_zeros | cardinal_with_leading_zeros + final_graph = graph_without_leading_zeros | cardinal_with_leading_zeros | cardinal_with_separators optional_minus_graph = pynini.closure(pynutil.insert("negative: ") + pynini.cross("-", "\"true\" "), 0, 1) + # --- Centralized logic for Address & Serial classes --- + # 1-3 digit groups read as cardinals, 4+ digits read digit-by-digit + limited_cardinal_graph = (self.digit | self.zero | self.teens_and_ties | self.graph_hundreds).optimize() + + any_digit = pynini.union( + NEMO_DIGIT, + pynini.project( + pynini.union( + pynini.string_file(get_abs_path("data/numbers/digit.tsv")), + pynini.string_file(get_abs_path("data/numbers/zero.tsv")), + ), + "input", + ), + ).optimize() + + digitwise_4plus = pynini.compose(any_digit**4 + pynini.closure(any_digit), self.single_digits_graph).optimize() + + self.code_num_graph = (limited_cardinal_graph | digitwise_4plus).optimize() + self.final_graph = final_graph.optimize() final_graph = optional_minus_graph + pynutil.insert("integer: \"") + self.final_graph + pynutil.insert("\"") final_graph = self.add_tokens(final_graph) diff --git a/nemo_text_processing/text_normalization/hi/taggers/date.py b/nemo_text_processing/text_normalization/hi/taggers/date.py index da917f3de..6f43e5d6e 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/date.py +++ b/nemo_text_processing/text_normalization/hi/taggers/date.py @@ -33,23 +33,31 @@ teens_ties = pynini.union(teens_ties_hi, teens_ties_en) teens_and_ties = pynutil.add_weight(teens_ties, -0.1) -# Read suffixes from file into a list with open(get_abs_path("data/date/suffixes.tsv"), "r", encoding="utf-8") as f: - suffixes_list = f.read().splitlines() + suffix_union = pynini.string_map([line.rstrip("\n") for line in f if line.strip()]) + with open(get_abs_path("data/date/prefixes.tsv"), "r", encoding="utf-8") as f: - prefixes_list = f.read().splitlines() + prefix_union = pynini.string_map([line.rstrip("\n") for line in f if line.strip()]) + +verbalized_hundreds = teens_ties_hi.project("output") +verbalized_unit = pynini.union(verbalized_hundreds, digit.project("output")) + +verbalized_year_sou = ( + verbalized_hundreds + pynini.accep(" सौ") + pynini.closure(pynini.accep(" ") + verbalized_unit, 0, 1) +) -# Create union of suffixes and prefixes -suffix_union = pynini.union(*suffixes_list) -prefix_union = pynini.union(*prefixes_list) +pad_latin = pynini.union(*[pynini.cross(str(i), f"0{i}") for i in range(1, 10)]) +pad_devanagari = pynini.union(*[pynini.cross(d, f"०{d}") for d in "१२३४५६७८९"]) class DateFst(GraphFst): """ Finite state transducer for classifying date, e.g. "०१-०४-२०२४" -> date { day: "एक" month: "अप्रैल" year: "दो हज़ार चौबीस" } - "०४-०१-२०२४" -> date { month: "अप्रैल" day: "एक" year: "दो हज़ार चौबीस" } - + "६ मार्च, २०१०" -> date { day: "छह" month: "मार्च" year: "दो हज़ार दस" } + "३१ मई, १९९० ई." -> date { day: "इकतीस" month: "मई" year: "उन्नीस सौ नब्बे" era: "ईसवी" } + "उन्नीस सौ बीस में" -> date { era: "उन्नीस सौ बीस में" } + "02-07-1970" -> date { day: "दो" month: "जुलाई" year: "उन्नीस सौ सत्तर" } Args: cardinal: cardinal GraphFst @@ -68,52 +76,137 @@ def __init__(self, cardinal: GraphFst): ) cardinal_graph = pynini.union( - digit, teens_and_ties, cardinal.graph_hundreds, graph_year_thousands, graph_year_hundreds_as_thousands + digit, + teens_and_ties, + cardinal.graph_hundreds, + graph_year_thousands, + graph_year_hundreds_as_thousands, ) graph_year = pynini.union(graph_year_thousands, graph_year_hundreds_as_thousands) + graph_year_era = pynini.union( + graph_year_thousands, + graph_year_hundreds_as_thousands, + cardinal.graph_hundreds, + ) + delete_dash = pynutil.delete("-") - delete_slash = pynutil.delete("/") + delete_comma = pynutil.delete(",") + delete_space = pynutil.delete(" ") + delete_optional_space = pynini.closure(pynutil.delete(" "), 0, 1) + delete_comma_sep = delete_comma + delete_optional_space + + day_num_padded = pynini.union( + days, + teens_and_ties, + ) - days_graph = pynutil.insert("day: \"") + days + pynutil.insert("\"") + insert_space + day_num_bare = pynini.union( + pynini.compose(pad_latin, days), + pynini.compose(pad_devanagari, days), + ) - months_graph = pynutil.insert("month: \"") + months + pynutil.insert("\"") + insert_space + days_graph_padded = pynutil.insert("day: \"") + day_num_padded + pynutil.insert("\"") + insert_space + days_graph_bare = pynutil.insert("day: \"") + day_num_bare + pynutil.insert("\"") + insert_space - years_graph = pynutil.insert("year: \"") + graph_year + pynutil.insert("\"") + insert_space + month_name_acceptor = pynini.project(months, "output") + + months_numeric_padded = months + + months_numeric_bare = pynini.union( + pynini.compose(pad_latin, months), + pynini.compose(pad_devanagari, months), + ) - graph_dd_mm = days_graph + delete_dash + months_graph + months_graph_numeric_padded = ( + pynutil.insert("month: \"") + months_numeric_padded + pynutil.insert("\"") + insert_space + ) + + months_fst_padded = pynini.union(months_numeric_padded, month_name_acceptor) + months_graph_padded = pynutil.insert("month: \"") + months_fst_padded + pynutil.insert("\"") + insert_space - graph_mm_dd = months_graph + delete_dash + days_graph + months_fst_bare = pynini.union(months_numeric_bare, month_name_acceptor) + months_graph_bare = pynutil.insert("month: \"") + months_fst_bare + pynutil.insert("\"") + insert_space - graph_mm_dd += pynutil.insert(" preserve_order: true ") + month_name_graph = pynutil.insert("month: \"") + month_name_acceptor + pynutil.insert("\"") + insert_space + + years_graph = pynutil.insert("year: \"") + graph_year + pynutil.insert("\"") + insert_space - # Graph for era era_graph = pynutil.insert("era: \"") + year_suffix + pynutil.insert("\"") + insert_space range_graph = pynini.cross("-", "से") - # Graph for year century_number = pynini.compose(pynini.closure(NEMO_ALL_DIGIT, 1), cardinal_graph) + pynini.accep("वीं") century_text = pynutil.insert("era: \"") + century_number + pynutil.insert("\"") + insert_space - # Updated logic to use suffix_union year_number = graph_year + suffix_union year_text = pynutil.insert("era: \"") + year_number + pynutil.insert("\"") + insert_space - # Updated logic to use prefix_union - year_prefix = pynutil.insert("era: \"") + prefix_union + insert_space + graph_year + pynutil.insert("\"") + year_prefix = pynutil.insert("era: \"") + prefix_union + pynini.accep(" ") + graph_year + pynutil.insert("\"") - delete_separator = pynini.union(delete_dash, delete_slash) - graph_dd_mm_yyyy = days_graph + delete_separator + months_graph + delete_separator + years_graph + year_prefix_suffix = ( + pynutil.insert("era: \"") + + prefix_union + + pynini.accep(" ") + + graph_year + + suffix_union + + pynutil.insert("\"") + ) - graph_mm_dd_yyyy = months_graph + delete_separator + days_graph + delete_separator + years_graph + graph_verbalized_year_suffix = ( + pynutil.insert("era: \"") + verbalized_year_sou + suffix_union + pynutil.insert("\"") + insert_space + ) - graph_mm_dd_yyyy += pynutil.insert(" preserve_order: true ") + graph_verbalized_year_bare = ( + pynutil.insert("era: \"") + verbalized_year_sou + pynutil.insert("\"") + insert_space + ) - graph_mm_yyyy = months_graph + delete_dash + insert_space + years_graph + graph_verbalized_year_prefix = ( + pynutil.insert("era: \"") + prefix_union + pynini.accep(" ") + verbalized_year_sou + pynutil.insert("\"") + ) - graph_year_suffix = era_graph + graph_verbalized_year_prefix_suffix = ( + pynutil.insert("era: \"") + + prefix_union + + pynini.accep(" ") + + verbalized_year_sou + + suffix_union + + pynutil.insert("\"") + ) + + graph_dd_mm = days_graph_padded + delete_dash + months_graph_padded + + graph_d_m = days_graph_bare + delete_dash + months_graph_bare + + graph_dd_mm_yyyy = days_graph_padded + delete_dash + months_graph_padded + delete_dash + years_graph + + graph_d_m_yyyy = days_graph_bare + delete_dash + months_graph_bare + delete_dash + years_graph + + graph_dd_month = days_graph_padded + delete_space + months_graph_numeric_padded + + graph_dd_month_comma_yyyy = ( + days_graph_padded + delete_space + months_graph_padded + delete_comma_sep + years_graph + ) + + graph_dd_month_comma_yyyy_era = ( + days_graph_padded + delete_space + months_graph_padded + delete_comma_sep + years_graph + era_graph + ) + + graph_month_comma_yyyy = months_graph_padded + delete_comma_sep + years_graph + + graph_month_comma_yyyy_era = months_graph_padded + delete_comma_sep + years_graph + era_graph + + graph_month_name_yyyy = month_name_graph + delete_space + years_graph + + graph_year_era_only = ( + pynutil.insert("era: \"") + + graph_year_era + + insert_space + + year_suffix + + pynutil.insert("\"") + + insert_space + ) graph_range = ( pynutil.insert("era: \"") @@ -126,21 +219,31 @@ def __init__(self, cardinal: GraphFst): + pynutil.insert(" preserve_order: true ") ) - # default assume dd_mm_yyyy + graph_year_suffix = era_graph final_graph = ( - pynutil.add_weight(graph_dd_mm, -0.001) - | graph_mm_dd + pynutil.add_weight(graph_dd_month_comma_yyyy_era, -0.003) + | pynutil.add_weight(graph_month_comma_yyyy_era, -0.003) | pynutil.add_weight(graph_dd_mm_yyyy, -0.001) - | graph_mm_dd_yyyy - | pynutil.add_weight(graph_mm_yyyy, -0.2) - | pynutil.add_weight(graph_year_suffix, -0.001) + | pynutil.add_weight(graph_d_m_yyyy, -0.001) + | pynutil.add_weight(graph_dd_month_comma_yyyy, -0.001) + | pynutil.add_weight(graph_dd_mm, -0.001) + | pynutil.add_weight(graph_d_m, -0.001) + | pynutil.add_weight(graph_dd_month, -0.001) + | pynutil.add_weight(graph_month_name_yyyy, -0.2) + | pynutil.add_weight(graph_month_comma_yyyy, -0.2) + | pynutil.add_weight(graph_year_era_only, -0.005) | pynutil.add_weight(graph_range, -0.005) + | pynutil.add_weight(graph_year_suffix, -0.001) | pynutil.add_weight(century_text, -0.001) - | pynutil.add_weight(year_text, -0.001) + | pynutil.add_weight(graph_verbalized_year_prefix_suffix, -0.012) + | pynutil.add_weight(graph_verbalized_year_prefix, -0.011) + | pynutil.add_weight(graph_verbalized_year_suffix, -0.010) + | pynutil.add_weight(graph_verbalized_year_bare, -0.009) + | pynutil.add_weight(year_prefix_suffix, -0.010) | pynutil.add_weight(year_prefix, -0.009) + | pynutil.add_weight(year_text, -0.001) ) self.final_graph = final_graph.optimize() - self.fst = self.add_tokens(self.final_graph) diff --git a/nemo_text_processing/text_normalization/hi/taggers/electronic.py b/nemo_text_processing/text_normalization/hi/taggers/electronic.py index 7807117e6..e1b93835e 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/electronic.py +++ b/nemo_text_processing/text_normalization/hi/taggers/electronic.py @@ -22,11 +22,11 @@ class ElectronicFst(GraphFst): """ Finite state transducer for classifying electronic: as URLs, email addresses, file paths, - IP addresses, domains, chemical formulas, and alphanumeric codes. + IP addresses, domains, and chemical formulas. e.g. kumar@gmail.com -> tokens { electronic { username: "kumar" domain: "gmail.com" } } e.g. https://google.com/ -> tokens { electronic { protocol: "https" domain: "google.com/" } } e.g. C:\\Users\\HP\\Desktop -> tokens { electronic { path: "C:\\Users\\HP\\Desktop" } } - e.g. 192.168.1.1 -> tokens { electronic { ip: "192.168.1.1" } } + e.g. 192.168.1.1 -> tokens { electronic { domain: "192.168.1.1" } } """ @@ -36,11 +36,31 @@ def __init__(self, deterministic: bool = True): subscript_digit = pynini.project( pynini.string_file(get_abs_path("data/electronic/subscript_digit.tsv")), "input" ) - alphanumeric = NEMO_ALPHA | NEMO_DIGIT | NEMO_HI_DIGIT | subscript_digit - # email - username_chars = NEMO_ALPHA | NEMO_DIGIT | pynini.accep(".") | pynini.accep("-") | pynini.accep("_") + symbol_dict = {"email": [], "url": [], "unix": [], "windows": [], "chem": []} + + with open(get_abs_path("data/electronic/symbol_classes.tsv"), "r", encoding="utf-8") as f: + for line in f: + if not line.strip(): + continue + parts = line.strip().split("\t") + if len(parts) == 2: + sym = parts[0] + classes = parts[1].split(",") + for c in classes: + if c in symbol_dict: + symbol_dict[c].append(sym) + + email_symbols = pynini.union(*symbol_dict["email"]) + url_symbols = pynini.union(*symbol_dict["url"]) + unix_symbols = pynini.union(*symbol_dict["unix"]) + win_symbols = pynini.union(*symbol_dict["windows"]) + chemical_symbols = pynini.union(*symbol_dict["chem"]) + + unix_segment_syms = pynini.union(*[s for s in symbol_dict["unix"] if s != "/"]) + + username_chars = NEMO_ALPHA | NEMO_DIGIT | email_symbols username = pynutil.insert("username: \"") + pynini.closure(username_chars, 1) + pynutil.insert("\"") domain_chars = NEMO_ALPHA | NEMO_DIGIT | pynini.accep(".") | pynini.accep("-") @@ -48,7 +68,6 @@ def __init__(self, deterministic: bool = True): email_graph = username + pynini.cross("@", "") + domain - # url: protocol handling for https://, http://, www., and combined forms protocol_start = pynini.cross("https://", "https") | pynini.cross("http://", "http") protocol_end = pynini.cross("www.", "www") protocol = ( @@ -61,36 +80,13 @@ def __init__(self, deterministic: bool = True): + pynutil.insert("\"") ) - url_path_chars = alphanumeric | pynini.union( - pynini.accep("."), - pynini.accep("-"), - pynini.accep("_"), - pynini.accep("/"), - pynini.accep("#"), - pynini.accep("?"), - pynini.accep("&"), - pynini.accep("="), - pynini.accep("%"), - pynini.accep("+"), - pynini.accep(":"), - ) + url_path_chars = alphanumeric | url_symbols url_path = pynini.closure(url_path_chars, 1) - url_domain = pynutil.insert(" domain: \"") + url_path + pynutil.insert("\"") - url_graph = protocol + url_domain - # file paths: Windows (C:\...), Unix (/...), and backslash-prefixed (\...) drive_letter = NEMO_ALPHA - windows_path_chars = alphanumeric | pynini.union( - pynini.accep("\\"), - pynini.accep("."), - pynini.accep("-"), - pynini.accep("_"), - pynini.accep(" "), - pynini.accep("("), - pynini.accep(")"), - ) + windows_path_chars = alphanumeric | win_symbols | pynini.accep(" ") windows_path = ( pynutil.insert("path: \"") + drive_letter @@ -100,24 +96,16 @@ def __init__(self, deterministic: bool = True): + pynutil.insert("\"") ) - unix_path_chars = alphanumeric | pynini.union( - pynini.accep("/"), - pynini.accep("."), - pynini.accep("-"), - pynini.accep("_"), - pynini.accep("$"), - ) - unix_path = ( - pynutil.insert("path: \"") + pynini.accep("/") + pynini.closure(unix_path_chars, 1) + pynutil.insert("\"") - ) + unix_path_chars = alphanumeric | unix_symbols + unix_segment_chars = alphanumeric | unix_segment_syms + unix_segment = pynini.closure(unix_segment_chars, 1) - backslash_path_chars = alphanumeric | pynini.union( - pynini.accep("\\"), - pynini.accep("."), - pynini.accep("-"), - pynini.accep("_"), - pynini.accep(" "), - ) + abs_unix_path = pynini.accep("/") + pynini.closure(unix_path_chars, 1) + rel_unix_path = unix_segment + pynini.accep("/") + pynini.closure(unix_path_chars, 0) + + unix_path = pynutil.insert("path: \"") + (abs_unix_path | rel_unix_path) + pynutil.insert("\"") + + backslash_path_chars = alphanumeric | unix_segment_syms | pynini.accep("\\") | pynini.accep(" ") backslash_path = ( pynutil.insert("path: \"") + pynini.accep("\\") @@ -125,12 +113,10 @@ def __init__(self, deterministic: bool = True): + pynutil.insert("\"") ) - # ip addresses: exactly 4 dot-separated octets ip_octet = pynini.closure(NEMO_DIGIT, 1, 3) dot_octet = pynini.accep(".") + ip_octet ip_address = pynutil.insert("domain: \"") + ip_octet + pynini.closure(dot_octet, 3, 3) + pynutil.insert("\"") - # domains: simple TLD-based (abc.com) and government/education suffixes (.gov.in, .ac.in) domain_segment_chars = NEMO_ALPHA | NEMO_DIGIT | pynini.accep("-") domain_segment = pynini.closure(domain_segment_chars, 1) @@ -144,7 +130,6 @@ def __init__(self, deterministic: bool = True): pynutil.insert("domain: \"") + domain_body + pynini.closure(pynini.accep("/"), 0, 1) + pynutil.insert("\"") ) - # file extensions: e.g. report.pdf, data.csv known_extensions = pynini.project( pynini.string_file(get_abs_path("data/electronic/file_extensions.tsv")), "input" ) @@ -154,27 +139,32 @@ def __init__(self, deterministic: bool = True): pynutil.insert("domain: \"") + filename_stem + pynini.accep(".") + known_extensions + pynutil.insert("\"") ) - # chemical formulas with subscript digits: e.g. H₂O, CO₂ - chemical_chars = NEMO_ALPHA | subscript_digit - chemical_formula = ( - pynutil.insert("domain: \"") + NEMO_ALPHA + pynini.closure(chemical_chars, 1) + pynutil.insert("\"") - ) + elements = pynini.project(pynini.string_file(get_abs_path("data/electronic/elements.tsv")), "input") + + chem_number = pynini.closure(NEMO_DIGIT | subscript_digit, 1) + + chem_block = elements + pynini.closure(chem_number, 0, 1) + + chem_sequence_chars = chem_block | chemical_symbols | chem_number + + raw_chemical = pynini.closure(chemical_symbols) + chem_block + pynini.closure(chem_sequence_chars) + + any_chem = pynini.closure(chem_sequence_chars) + has_open = any_chem + pynini.accep("(") + any_chem + no_open = pynini.difference(any_chem, has_open) + ends_with_close = any_chem + pynini.accep(")") - # alphanumeric codes: strings containing both letters and digits, - # optionally separated by hyphens, e.g. IELF004, N95, GSAT-18, F-35B - alnum_seg = pynini.closure(NEMO_ALPHA | NEMO_DIGIT, 1) - alphanumeric_pattern = alnum_seg + pynini.closure(pynini.accep("-") + alnum_seg) + unbalanced_trailing = pynini.intersect(no_open, ends_with_close) + valid_chemical = pynini.difference(raw_chemical, unbalanced_trailing).optimize() - alnum_hyp_sigma = pynini.closure(NEMO_ALPHA | NEMO_DIGIT | pynini.accep("-")) - contains_alpha = alnum_hyp_sigma + NEMO_ALPHA + alnum_hyp_sigma - contains_digit = alnum_hyp_sigma + NEMO_DIGIT + alnum_hyp_sigma - alphanumeric_code_fst = pynini.intersect( - pynini.intersect(alphanumeric_pattern, contains_alpha), contains_digit - ).optimize() + # Recognise a chemical formula only when it uses subscript notation + chem_sigma = pynini.closure(NEMO_ALPHA | NEMO_DIGIT | subscript_digit | chemical_symbols) + contains_subscript = chem_sigma + subscript_digit + chem_sigma + valid_chemical = pynini.intersect(valid_chemical, contains_subscript).optimize() - alphanumeric_code = pynutil.insert("domain: \"") + alphanumeric_code_fst + pynutil.insert("\"") + # Chemical formulas carry a dedicated tag so the verbalizer can spell element + chemical_formula = pynutil.insert("fragment_id: \"") + valid_chemical + pynutil.insert("\"") - # Weights use 3 tiers: structurally unambiguous (1.0), moderately general (1.1), greedy (1.2) graph = ( pynutil.add_weight(url_graph, 1.0) | pynutil.add_weight(email_graph, 1.0) @@ -185,7 +175,6 @@ def __init__(self, deterministic: bool = True): | pynutil.add_weight(combined_domain, 1.1) | pynutil.add_weight(file_with_extension, 1.1) | pynutil.add_weight(chemical_formula, 1.2) - | pynutil.add_weight(alphanumeric_code, 1.2) ) self.graph = graph.optimize() diff --git a/nemo_text_processing/text_normalization/hi/taggers/fraction.py b/nemo_text_processing/text_normalization/hi/taggers/fraction.py index b5528deba..8b72b25b2 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/fraction.py +++ b/nemo_text_processing/text_normalization/hi/taggers/fraction.py @@ -26,18 +26,18 @@ ) from nemo_text_processing.text_normalization.hi.utils import get_abs_path -HI_ONE_HALF = "१/२" # 1/2 -HI_ONE_QUARTER = "१/४" # 1/4 -HI_THREE_QUARTERS = "३/४" # 3/4 +HI_ONE_HALF = "१/२" +HI_ONE_QUARTER = "१/४" +HI_THREE_QUARTERS = "३/४" class FractionFst(GraphFst): """ Finite state transducer for classifying fraction "२३ ४/६" -> - fraction { integer: "तेईस" numerator: "चार" denominator: "छः"} + fraction { integer: "तेईस" numerator: "चार" denominator: "छह"} ४/६" -> - fraction { numerator: "चार" denominator: "छः"} + fraction { numerator: "चार" denominator: "छह"} Args: @@ -54,13 +54,16 @@ def __init__(self, cardinal, deterministic: bool = True): self.optional_graph_negative = pynini.closure( pynutil.insert("negative: ") + pynini.cross("-", "\"true\"") + pynutil.insert(NEMO_SPACE), 0, 1 ) + self.integer = pynutil.insert("integer_part: \"") + cardinal_graph + pynutil.insert("\"") + self.numerator = ( pynutil.insert("numerator: \"") + cardinal_graph + pynini.cross(pynini.union("/", NEMO_SPACE + "/" + NEMO_SPACE), "\"") + pynutil.insert(NEMO_SPACE) ) + self.denominator = pynutil.insert("denominator: \"") + cardinal_graph + pynutil.insert("\"") dedh_dhai_graph = pynini.string_map( @@ -77,6 +80,15 @@ def __init__(self, cardinal, deterministic: bool = True): paune_numbers = paune + pynini.cross(NEMO_SPACE + HI_THREE_QUARTERS, "") paune_graph = pynutil.insert(HI_PAUNE) + pynutil.insert(NEMO_SPACE) + paune_numbers + common_fraction_map = pynini.string_file(get_abs_path("data/fraction/common_fractions.tsv")) + + graph_common_fraction = ( + pynutil.insert("morphosyntactic_features: \"") + + common_fraction_map + + pynutil.insert("\"") + + pynutil.insert(NEMO_SPACE) + ) + graph_dedh_dhai = ( pynutil.insert("morphosyntactic_features: \"") + dedh_dhai_graph @@ -114,10 +126,11 @@ def __init__(self, cardinal, deterministic: bool = True): weighted_graph = ( final_graph + | pynutil.add_weight(graph_common_fraction, -0.3) | pynutil.add_weight(graph_dedh_dhai, -0.2) + | pynutil.add_weight(graph_paune, -0.2) | pynutil.add_weight(graph_savva, -0.1) | pynutil.add_weight(graph_sadhe, -0.1) - | pynutil.add_weight(graph_paune, -0.2) ) self.graph = weighted_graph diff --git a/nemo_text_processing/text_normalization/hi/taggers/measure.py b/nemo_text_processing/text_normalization/hi/taggers/measure.py index e18111696..67043b727 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/measure.py +++ b/nemo_text_processing/text_normalization/hi/taggers/measure.py @@ -28,8 +28,8 @@ HI_SADHE, HI_SAVVA, HYPHEN, - INPUT_LOWER_CASED, LOWERCASE_X, + MIN_NEG_WEIGHT, NEMO_CHAR, NEMO_DIGIT, NEMO_HI_DIGIT, @@ -50,11 +50,15 @@ from nemo_text_processing.text_normalization.hi.utils import get_abs_path digit = pynini.string_file(get_abs_path("data/numbers/digit.tsv")) -# Load both Hindi (Devanagari) and English (Arabic) number mappings -teens_ties_hi = pynini.string_file(get_abs_path("data/numbers/teens_and_ties.tsv")) -teens_ties_en = pynini.string_file(get_abs_path("data/numbers/teens_and_ties_en.tsv")) -teens_ties = pynini.union(teens_ties_hi, teens_ties_en) -teens_and_ties = pynutil.add_weight(teens_ties, -0.1) + +# Shared Address Maps +zero = pynini.string_file(get_abs_path("data/numbers/zero.tsv")) +telephone_number = pynini.string_file(get_abs_path("data/telephone/number.tsv")) +states_map = pynini.string_file(get_abs_path("data/address/states.tsv")) +cities_map = pynini.string_file(get_abs_path("data/address/cities.tsv")) +special_characters_map = pynini.string_file(get_abs_path("data/address/special_characters.tsv")) +letters_map = pynini.string_file(get_abs_path("data/address/letters.tsv")) +context_map = pynini.string_file(get_abs_path("data/address/context.tsv")) class MeasureFst(GraphFst): @@ -71,32 +75,27 @@ class MeasureFst(GraphFst): for False multiple transduction are generated (used for audio-based normalization) """ - def get_structured_address_graph(self, ordinal: GraphFst, input_case: str): + def get_structured_address_graph(self, cardinal: GraphFst, ordinal: GraphFst, input_case: str): """ Minimal address tagger for state/city + pincode patterns only. - Highly optimized for performance. Examples: "मुंबई ८८४४०४" -> "मुंबई आठ आठ चार चार शून्य चार" "गोवा १२३४५६" -> "गोवा एक दो तीन चार पाँच छह" + "100 फीट रोड, चेन्नई" -> "एक सौ फीट रोड, चेन्नई" """ # State/city keywords - states = pynini.string_file(get_abs_path("data/address/states.tsv")) - cities = pynini.string_file(get_abs_path("data/address/cities.tsv")) - state_city_names = pynini.union(states, cities).optimize() - - # Digit mappings - num_token = ( - digit - | pynini.string_file(get_abs_path("data/numbers/zero.tsv")) - | pynini.string_file(get_abs_path("data/telephone/number.tsv")) - ).optimize() - # Pincode (6 digits) + state_city_names = pynini.union(states_map, cities_map).optimize() + + # Digit mappings (shared maps loaded once at module level) + num_token = (digit | zero | telephone_number).optimize() + + # Pincode (6 digits) -> always digit-by-digit (length >= 4) pincode = (num_token + pynini.closure(insert_space + num_token, 5, 5)).optimize() - # Street number (1-4 digits) - street_num = (num_token + pynini.closure(insert_space + num_token, 0, 3)).optimize() + # Street number: Use centralized digit-by-digit logic from cardinal + street_num = cardinal.code_num_graph # Text: words with trailing separator (comma? + space) any_digit = pynini.union(NEMO_HI_DIGIT, NEMO_DIGIT).optimize() @@ -107,7 +106,9 @@ def get_structured_address_graph(self, ordinal: GraphFst, input_case: str): # Separator: optional comma followed by mandatory space sep = pynini.closure(pynini.accep(COMMA), 0, 1) + pynini.accep(NEMO_SPACE) word_with_sep = word + sep - text = pynini.closure(word_with_sep, 0, 5).optimize() + # Consume inline address numbers using the same 1-3 (cardinal) / 4+ (digit-by-digit) rule + num_with_sep = street_num + sep + text = pynini.closure(pynini.union(word_with_sep, num_with_sep), 0, 5).optimize() # Pattern: [street_num + sep]? text state/city [space pincode] pattern = ( @@ -124,34 +125,26 @@ def get_structured_address_graph(self, ordinal: GraphFst, input_case: str): ) return pynutil.add_weight(graph, 1.0).optimize() - def get_address_graph(self, ordinal: GraphFst, input_case: str): + def get_address_graph(self, cardinal: GraphFst, ordinal: GraphFst, serial: GraphFst, input_case: str): """ - Address tagger that converts digits/hyphens/slashes character-by-character - when address context keywords are present. - English words and ordinals are converted to Hindi transliterations. + Address tagger that fires when address context keywords are present. Examples: - "७०० ओक स्ट्रीट" -> "सात शून्य शून्य ओक स्ट्रीट" - "६६-४ पार्क रोड" -> "छह छह हाइफ़न चार पार्क रोड" + "७०० ओक स्ट्रीट" -> "सात सौ ओक स्ट्रीट" + "६६-४ पार्क रोड" -> "छियासठ हाइफ़न चार पार्क रोड" + "593988" (6-digit pincode) -> "पाँच नौ तीन नौ आठ आठ" + "32A नाज़ प्लाज़ा" -> "बत्तीस ए नाज़ प्लाज़ा" """ + # Retain internal weights of ordinal graph ordinal_graph = ordinal.graph # Alphanumeric to word mappings (digits, special characters, telephone digits) - char_to_word = ( - digit - | pynini.string_file(get_abs_path("data/numbers/zero.tsv")) - | pynini.string_file(get_abs_path("data/address/special_characters.tsv")) - | pynini.string_file(get_abs_path("data/telephone/number.tsv")) + char_to_word = (digit | zero | special_characters_map | telephone_number).optimize() + letter_to_word = capitalized_input_graph(letters_map) + # Identity acceptor for keywords (Devanagari/English) to prevent unintended rewrites/transliteration + address_keywords = pynini.project( + capitalized_input_graph(context_map), + "input", ).optimize() - letter_to_word = pynini.string_file(get_abs_path("data/address/letters.tsv")) - letter_to_word = capitalized_input_graph(letter_to_word) - address_keywords_hi = pynini.string_file(get_abs_path("data/address/context.tsv")) - - # English address keywords with Hindi translation (case-insensitive) - en_to_hi_map = pynini.string_file(get_abs_path("data/address/en_to_hi_mapping.tsv")) - if input_case != INPUT_LOWER_CASED: - en_to_hi_map = capitalized_input_graph(en_to_hi_map) - address_keywords_en = pynini.project(en_to_hi_map, "input") - address_keywords = pynini.union(address_keywords_hi, address_keywords_en) # Alphanumeric processing: treat digits, letters, and -/ as convertible tokens single_digit = pynini.union(NEMO_DIGIT, NEMO_HI_DIGIT).optimize() @@ -162,24 +155,34 @@ def get_address_graph(self, ordinal: GraphFst, input_case: str): NEMO_CHAR, pynini.union(NEMO_WHITE_SPACE, convertible_char, pynini.accep(COMMA)) ).optimize() - # Token processors with weights: prefer ordinals and known English→Hindi words - # Delete space before comma to avoid Sparrowhawk "sil" issue - comma_processor = pynutil.add_weight(delete_space + pynini.accep(COMMA), 0.0) - ordinal_processor = pynutil.add_weight(insert_space + ordinal_graph, -5.0) - english_word_processor = pynutil.add_weight(insert_space + en_to_hi_map, -3.0) - letter_processor = pynutil.add_weight(insert_space + pynini.compose(single_letter, letter_to_word), 0.5) - digit_char_processor = pynutil.add_weight(insert_space + pynini.compose(convertible_char, char_to_word), 0.0) - other_word_processor = pynutil.add_weight(insert_space + pynini.closure(non_space_char, 1), 0.1) + comma_processor = delete_space + pynini.accep(COMMA) + ordinal_processor = insert_space + ordinal_graph + latin_word = single_letter + pynini.closure(single_letter, 1) + english_word_processor = insert_space + latin_word + letter_processor = insert_space + pynini.compose(single_letter, letter_to_word) + special_char_processor = insert_space + pynini.compose(special_chars, char_to_word) + other_word_processor = insert_space + pynini.closure(non_space_char, 1) + + code_processor = insert_space + serial.mixed_alphanum_graph + + # Pure numeric runs: 1-3 digits read as cardinal, 4+ digits read digit-by-digit + number_run = pynini.compose(pynini.closure(single_digit, 1), cardinal.code_num_graph).optimize() + + # A tiny positive penalty prevents multiple uses of this arc, forcing the FST to consume contiguous digits as ONE run instead of aggressively splitting them. + number_run_processor = insert_space + number_run token_processor = ( - ordinal_processor - | english_word_processor - | letter_processor - | digit_char_processor - | pynini.accep(NEMO_SPACE) + pynini.accep(NEMO_SPACE) | comma_processor - | other_word_processor + | special_char_processor + | code_processor + | pynutil.add_weight(ordinal_processor, MIN_NEG_WEIGHT) + | pynutil.add_weight(number_run_processor, 0.5) # Keeps numbers together + | letter_processor + | english_word_processor + | pynutil.add_weight(other_word_processor, 0.1) # Keeps Hindi words together ).optimize() + full_string_processor = pynini.closure(token_processor, 1).optimize() # Window-based context matching around address keywords for robust detection @@ -201,9 +204,9 @@ def get_address_graph(self, ordinal: GraphFst, input_case: str): + address_graph + pynutil.insert('" } preserve_order: true') ) - return pynutil.add_weight(graph, 1.05).optimize() + return graph.optimize() - def __init__(self, cardinal: GraphFst, decimal: GraphFst, ordinal: GraphFst, input_case: str): + def __init__(self, cardinal: GraphFst, decimal: GraphFst, ordinal: GraphFst, serial: GraphFst, input_case: str): super().__init__(name="measure", kind="classify") cardinal_graph = ( @@ -451,8 +454,8 @@ def __init__(self, cardinal: GraphFst, decimal: GraphFst, ordinal: GraphFst, inp + pynutil.insert("\"") ) - address_graph = self.get_address_graph(ordinal, input_case) - structured_address_graph = self.get_structured_address_graph(ordinal, input_case) + address_graph = self.get_address_graph(cardinal, ordinal, serial, input_case) + structured_address_graph = self.get_structured_address_graph(cardinal, ordinal, input_case) graph = ( pynutil.add_weight(graph_decimal, 0.1) diff --git a/nemo_text_processing/text_normalization/hi/taggers/money.py b/nemo_text_processing/text_normalization/hi/taggers/money.py index 01e46352f..16f389ae7 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/money.py +++ b/nemo_text_processing/text_normalization/hi/taggers/money.py @@ -16,18 +16,18 @@ from pynini.lib import pynutil from nemo_text_processing.text_normalization.hi.graph_utils import GraphFst, insert_space -from nemo_text_processing.text_normalization.hi.utils import get_abs_path +from nemo_text_processing.text_normalization.hi.utils import get_abs_path, load_labels currency_graph = pynini.string_file(get_abs_path("data/money/currency.tsv")) +currency_singular_graph = pynini.string_file(get_abs_path("data/money/currency_singular.tsv")) class MoneyFst(GraphFst): """ Finite state transducer for classifying money, suppletive aware, e.g. - ₹५० -> money { money { currency_maj: "रुपए" integer_part: "पचास" } - ₹५०.५० -> money { currency_maj: "रुपए" integer_part: "पचास" fractional_part: "पचास" currency_min: "centiles" } - ₹०.५० -> money { currency_maj: "रुपए" integer_part: "शून्य" fractional_part: "पचास" currency_min: "centiles" } - Note that the 'centiles' string is a placeholder to handle by the verbalizer by applying the corresponding minor currency denomination + ₹५० -> money { currency_maj: "रुपए" integer_part: "पचास" } + ₹५०.५० -> money { currency_maj: "रुपए" integer_part: "पचास" fractional_part: "पचास" currency_min: "पैसे" } + ₹०.५० -> { money { currency_maj: "रुपए" integer_part: "शून्य" fractional_part: "पचास" currency_min: "पैसे" } Args: cardinal: CardinalFst @@ -41,30 +41,118 @@ def __init__(self, cardinal: GraphFst): cardinal_graph = cardinal.final_graph + _en_to_hi_digit = pynini.string_file(get_abs_path("data/ordinal/en_to_hi_digit.tsv")) + _deva_to_ascii = pynini.invert(_en_to_hi_digit) + deva_to_ascii = pynini.closure(_deva_to_ascii | pynini.union(*"0123456789"), 1) + + _ascii_digit = pynini.union(*"0123456789") + _ascii_nonzero = pynini.union(*"123456789") + _deva_nonzero = pynini.union(*"१२३४५६७८९") + _any_digit = _ascii_digit | pynini.union(*"०१२३४५६७८९") + _any_nonzero = _ascii_nonzero | _deva_nonzero + optional_graph_negative = pynini.closure( - pynutil.insert("negative: ") + pynini.cross("-", "\"true\"") + insert_space, + pynutil.insert("negative: ") + pynini.cross("-", '"true"') + insert_space, 0, 1, ) + currency_major = pynutil.insert('currency_maj: "') + currency_graph + pynutil.insert('"') + currency_major_singular = pynutil.insert('currency_maj: "') + currency_singular_graph + pynutil.insert('"') + + one = pynini.union("1", "१") + integer_one = pynutil.insert('integer_part: "') + (one @ cardinal_graph) + pynutil.insert('"') integer = pynutil.insert('integer_part: "') + cardinal_graph + pynutil.insert('"') - fraction = pynutil.insert('fractional_part: "') + cardinal_graph + pynutil.insert('"') - currency_minor = pynutil.insert('currency_min: "') + pynutil.insert("centiles") + pynutil.insert('"') - graph_major_only = optional_graph_negative + currency_major + insert_space + integer - graph_major_and_minor = ( + strip_trailing_zeros = pynini.closure(_ascii_digit) + _ascii_nonzero + pynini.closure(pynutil.delete("0")) + canonicalise = ( + (pynutil.delete("0") + _ascii_nonzero) + | (_ascii_nonzero + pynutil.insert("0")) + | (_ascii_nonzero + _ascii_digit) + ) + two_digits_fractional_part = deva_to_ascii @ strip_trailing_zeros @ canonicalise + + fraction = ( + pynutil.insert('fractional_part: "') + (two_digits_fractional_part @ cardinal_graph) + pynutil.insert('"') + ) + + optional_delete_fractional_zeros = pynini.closure( + pynutil.delete(".") + pynini.closure(pynutil.delete("0") | pynutil.delete("०"), 1), + 0, + 1, + ) + + has_3plus_sig_digits = _any_digit + _any_digit + _any_nonzero + pynini.closure(_any_digit) + single_digit = _any_digit @ cardinal.single_digits_graph + decimal_digits = ( + pynutil.insert('fractional_part: "') + + single_digit + + pynini.closure(insert_space + single_digit) + + pynutil.insert('"') + ) + guarded_decimal_digits = has_3plus_sig_digits @ decimal_digits + + graph_decimal_path = ( optional_graph_negative + currency_major + insert_space - + integer + + pynutil.insert('integer_part: "') + + cardinal_graph + + pynutil.insert('"') + pynini.cross(".", " ") - + fraction + + guarded_decimal_digits + ).optimize() + + graph_major_only_singular = ( + optional_graph_negative + + currency_major_singular + insert_space - + currency_minor - ) + + integer_one + + optional_delete_fractional_zeros + ).optimize() + + graph_major_only = ( + optional_graph_negative + currency_major + insert_space + integer + optional_delete_fractional_zeros + ).optimize() + + maj_labels = load_labels(get_abs_path("data/money/currency.tsv")) + maj_singular_labels = load_labels(get_abs_path("data/money/currency_singular.tsv")) + maj_to_min = dict(load_labels(get_abs_path("data/money/major_minor_currencies.tsv"))) + + def _build_major_and_minor(sym_maj_labels, int_graph): + result = None + for sym, maj in sym_maj_labels: + min_name = maj_to_min.get(maj) + if not min_name: + continue - graph_currencies = graph_major_only | graph_major_and_minor + curr_maj = pynutil.insert('currency_maj: "') + pynini.cross(sym, maj) + pynutil.insert('"') + curr_min = pynutil.insert('currency_min: "') + pynutil.insert(min_name) + pynutil.insert('"') + + g = ( + optional_graph_negative + + curr_maj + + insert_space + + int_graph + + pynini.cross(".", " ") + + fraction + + insert_space + + curr_min + ).optimize() + + result = g if result is None else pynini.union(result, g).optimize() + + return result + + graph_major_and_minor = _build_major_and_minor(maj_labels, integer) + graph_major_and_minor_singular = _build_major_and_minor(maj_singular_labels, integer_one) + + graph_currencies = ( + pynutil.add_weight(graph_major_only_singular | graph_major_and_minor_singular, -0.001) + | pynutil.add_weight(graph_decimal_path, -0.0005) + | graph_major_only + | graph_major_and_minor + ) graph = graph_currencies.optimize() - final_graph = self.add_tokens(graph) - self.fst = final_graph + self.fst = self.add_tokens(graph) diff --git a/nemo_text_processing/text_normalization/hi/taggers/punctuation.py b/nemo_text_processing/text_normalization/hi/taggers/punctuation.py index 14c9a1a55..ea9cfc72e 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/punctuation.py +++ b/nemo_text_processing/text_normalization/hi/taggers/punctuation.py @@ -1,3 +1,17 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + import sys from unicodedata import category diff --git a/nemo_text_processing/text_normalization/hi/taggers/roman.py b/nemo_text_processing/text_normalization/hi/taggers/roman.py new file mode 100644 index 000000000..ea8d259fe --- /dev/null +++ b/nemo_text_processing/text_normalization/hi/taggers/roman.py @@ -0,0 +1,138 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.text_normalization.hi.graph_utils import GraphFst, convert_space, insert_space +from nemo_text_processing.text_normalization.hi.utils import get_abs_path, load_labels + + +class RomanFst(GraphFst): + """ + Finite state transducer for classifying Roman numerals in Hindi text. + e.g. भास्कर-II -> tokens { roman { key_cardinal: "भास्कर" integer: "II" } } + e.g. कक्षा XII -> tokens { roman { key_cardinal: "कक्षा" integer: "XII" } } + e.g. XIIवीं कक्षा -> tokens { roman { integer: "XII" default_ordinal: "बारहवीं" key_cardinal: "कक्षा" } } + e.g. IVथी कक्षा -> tokens { roman { integer: "IV" default_ordinal: "चौथी" key_cardinal: "कक्षा" } } + + Args: + deterministic: if True will provide a single transduction option, + for False multiple transduction are generated (used for audio-based normalization) + """ + + def __init__(self, deterministic: bool = True): + super().__init__(name="roman", kind="classify", deterministic=deterministic) + + roman_graph = pynini.string_file(get_abs_path("data/roman/roman_to_spoken.tsv")).optimize() + roman_numeral_only = pynini.project(roman_graph, "input").optimize() + + devanagari_chars = pynini.project( + pynini.string_file(get_abs_path("data/serial/chars.tsv")), "input" + ).optimize() + + devanagari_word = pynini.closure(devanagari_chars, 1).optimize() + + devanagari_phrase = ( + devanagari_word + pynini.closure((pynini.accep(" ") | pynini.accep("-")) + devanagari_word) + ).optimize() + + separator = (pynini.accep("-") | pynini.accep(" ")).optimize() + + key_before_numeral = ( + pynutil.insert("preserve_order: true ") + + pynutil.insert('key_cardinal: "') + + convert_space(devanagari_phrase) + + pynutil.insert('"') + + pynutil.delete(separator) + + insert_space + + pynutil.insert('integer: "') + + roman_numeral_only + + pynutil.insert('"') + ).optimize() + + numeral_before_key = ( + pynutil.insert("preserve_order: true ") + + pynutil.insert('integer: "') + + roman_numeral_only + + pynutil.insert('"') + + pynutil.delete(separator) + + insert_space + + pynutil.insert('key_cardinal: "') + + convert_space(devanagari_phrase) + + pynutil.insert('"') + ).optimize() + + roman_rows = load_labels(get_abs_path("data/roman/roman_to_spoken.tsv")) + numerals_by_len_desc = sorted((n for n, _ in roman_rows), key=len, reverse=True) + + exception_rows = load_labels(get_abs_path("data/roman/roman_ordinal_exceptions.tsv")) + exception_fused_set = {fused for fused, _ in exception_rows} + + suffix_rows_raw = load_labels(get_abs_path("data/ordinal/suffixes.tsv")) + load_labels( + get_abs_path("data/ordinal/suffixes_map.tsv") + ) + + exception_graphs = [] + for fused, spoken_word in exception_rows: + matched_numeral = next(c for c in numerals_by_len_desc if fused.startswith(c)) + exception_graphs.append( + pynutil.insert('integer: "' + matched_numeral + '"') + + insert_space + + pynutil.insert('default_ordinal: "' + spoken_word + '"') + + pynutil.delete(fused) + ) + glued_ordinal_exceptions_graph = pynini.union(*exception_graphs).optimize() + + regular_row_graphs = [] + for numeral, spoken in roman_rows: + for row in suffix_rows_raw: + + suffix_input = row[0] + suffix_output = row[1] if len(row) > 1 else row[0] + + fused = numeral + suffix_input + if fused in exception_fused_set: + continue + spoken_ordinal = spoken + suffix_output + regular_row_graphs.append( + pynutil.insert('integer: "' + numeral + '"') + + insert_space + + pynutil.insert('default_ordinal: "' + spoken_ordinal + '"') + + pynutil.delete(fused) + ) + glued_ordinal_regular_graph = pynini.union(*regular_row_graphs).optimize() + + roman_glued_ordinal_fields = pynini.union( + pynutil.add_weight(glued_ordinal_exceptions_graph, -0.1), + glued_ordinal_regular_graph, + ).optimize() + + roman_glued_ordinal = ( + pynutil.insert("preserve_order: true ") + + roman_glued_ordinal_fields + + pynini.closure( + pynutil.delete(" ") + + insert_space + + pynutil.insert('key_cardinal: "') + + convert_space(devanagari_phrase) + + pynutil.insert('"'), + 0, + 1, + ) + ).optimize() + + graph = pynini.union(key_before_numeral, numeral_before_key, roman_glued_ordinal).optimize() + + self.fst = self.add_tokens(graph).optimize() diff --git a/nemo_text_processing/text_normalization/hi/taggers/serial.py b/nemo_text_processing/text_normalization/hi/taggers/serial.py new file mode 100644 index 000000000..3f2244dd8 --- /dev/null +++ b/nemo_text_processing/text_normalization/hi/taggers/serial.py @@ -0,0 +1,162 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.text_normalization.hi.graph_utils import ( + NEMO_ALPHA, + NEMO_DIGIT, + NEMO_NOT_SPACE, + NEMO_SIGMA, + TO_LOWER, + GraphFst, + convert_space, +) +from nemo_text_processing.text_normalization.hi.utils import get_abs_path + + +class SerialFst(GraphFst): + """ + Finite state transducer for classifying serial strings in Hindi. + Handles Devanagari-numeric mixtures, complex delimited number chains, + symbols, and powers. Supports both ASCII (0-9) and Devanagari (०-९) digits. + + e.g. कोविड-19 -> tokens { name: "कोविड-उन्नीस" } + e.g. 5जी -> tokens { name: "पाँच जी" } + e.g. ३जी -> tokens { name: "तीन जी" } + e.g. 2^2 -> tokens { name: "दो स्क्वेर्ड" } + e.g. 2^4 -> tokens { name: "दो टु द पावर चार" } + e.g. 1-800-555 -> tokens { name: "एक-आठ सौ-पाँच सौ पचपन" } + e.g. B-60 -> tokens { name: "बी-साठ" } + e.g. A12 -> tokens { name: "ए बारह" } + e.g. FY2024 -> tokens { name: "एफ वाई दो शून्य दो चार" } + """ + + def __init__( + self, + cardinal: GraphFst, + deterministic: bool = True, + ): + super().__init__(name="serial", kind="classify", deterministic=deterministic) + + digit_graph = pynini.string_file(get_abs_path("data/numbers/digit.tsv")) + zero_graph = pynini.string_file(get_abs_path("data/numbers/zero.tsv")) + + devanagari_digits = pynini.project( + pynini.union(digit_graph, zero_graph), + "input", + ).optimize() + + any_digit = pynini.union(NEMO_DIGIT, devanagari_digits).optimize() + + # Fetch centralized 1-3 vs 4+ digit logic from cardinal + num_graph = cardinal.code_num_graph + + symbols_graph = pynini.string_file(get_abs_path("data/serial/special_symbols.tsv")).optimize() + + devanagari_chars = pynini.string_file(get_abs_path("data/serial/chars.tsv")).optimize() + + letter_graph = pynini.string_file(get_abs_path("data/address/letters.tsv")) + letter_graph = (letter_graph | pynini.compose(TO_LOWER, letter_graph)).optimize() + latin_letters = letter_graph + pynini.closure(pynutil.insert(" ") + letter_graph) + latin_letters = latin_letters.optimize() + + devanagari_word = pynini.closure(devanagari_chars, 2).optimize() + + delimiter = (pynini.accep("-") | pynini.accep("/") | pynini.accep(" ")).optimize() + + alphas = (latin_letters | devanagari_word).optimize() + segment = (alphas | num_graph | symbols_graph).optimize() + + serial_core = segment + pynini.closure(delimiter + segment, 1) + serial_core = serial_core.optimize() + + serial_graph = serial_core + + all_alphas = pynini.union(NEMO_ALPHA, devanagari_chars).optimize() + + insert_space_alpha_digit = pynini.cdrewrite(pynutil.insert(" "), all_alphas, any_digit, NEMO_SIGMA) + insert_space_digit_alpha = pynini.cdrewrite(pynutil.insert(" "), any_digit, all_alphas, NEMO_SIGMA) + space_inserter = pynini.compose(insert_space_alpha_digit, insert_space_digit_alpha).optimize() + + glued_serial = pynini.compose(space_inserter, serial_core).optimize() + serial_graph = pynini.union(serial_graph, glued_serial).optimize() + + # Reusable mixed alphanumeric-code graph + code_join_char = pynini.union(all_alphas, any_digit, pynini.accep("-"), pynini.accep("/")) + has_letter = pynini.closure(code_join_char) + all_alphas + pynini.closure(code_join_char) + has_digit = pynini.closure(code_join_char) + any_digit + pynini.closure(code_join_char) + mixed_code_only = pynini.intersect( + pynini.intersect(pynini.closure(code_join_char, 1), has_letter), has_digit + ).optimize() + + self.mixed_alphanum_graph = pynini.compose(mixed_code_only, serial_graph).optimize() + + power_special = pynutil.add_weight( + pynini.string_file(get_abs_path("data/serial/power_special.tsv")), -1.0 + ).optimize() + + power_generic = pynutil.add_weight( + (pynutil.delete("^") + pynutil.insert(" टु द पावर ") + num_graph), 1.0 + ).optimize() + + power_suffix = pynini.union(power_special, power_generic).optimize() + power_graph = num_graph + power_suffix + serial_graph = pynini.union(serial_graph, power_graph).optimize() + + serial_graph = pynini.compose(pynini.closure(NEMO_NOT_SPACE, 2), serial_graph).optimize() + + pure_word_slash = pynini.closure(NEMO_ALPHA, 1) + pynini.accep("/") + pynini.closure(NEMO_ALPHA, 1) + + letter_join_char = NEMO_ALPHA | pynini.accep("-") | pynini.accep("/") + contains_latin_letter = pynini.closure(letter_join_char) + NEMO_ALPHA + pynini.closure(letter_join_char) + pure_latin_word = pynini.intersect(pynini.closure(letter_join_char, 1), contains_latin_letter).optimize() + + dimension_pattern = ( + pynini.closure(any_digit, 1) + (pynini.accep("x") | pynini.accep("X")) + pynini.closure(any_digit, 1) + ) + + ordinal_suffixes = pynini.project( + pynini.union( + pynini.string_file(get_abs_path("data/ordinal/suffixes.tsv")), + pynini.string_file(get_abs_path("data/ordinal/suffixes_map.tsv")), + ), + "input", + ).optimize() + ordinal_pattern = pynini.closure(any_digit, 1) + ordinal_suffixes + + date_year_suffix = pynini.project( + pynini.string_file(get_abs_path("data/date/year_suffix.tsv")), + "input", + ).optimize() + date_suffixes = pynini.project( + pynini.string_file(get_abs_path("data/date/suffixes.tsv")), + "input", + ).optimize() + date_pattern = ( + pynini.closure(any_digit, 1) + + pynini.closure(pynini.accep("-") + pynini.closure(any_digit, 1), 0) + + pynini.accep(" ") + + pynini.union(date_year_suffix, date_suffixes) + ) + + exclusions = pure_word_slash | pure_latin_word | dimension_pattern | ordinal_pattern | date_pattern + accepted_inputs = pynini.difference(NEMO_SIGMA, exclusions).optimize() + + serial_graph = pynini.compose(accepted_inputs, serial_graph).optimize() + + self.graph = serial_graph.optimize() + graph = pynutil.insert('name: "') + convert_space(self.graph).optimize() + pynutil.insert('"') + self.fst = graph.optimize() diff --git a/nemo_text_processing/text_normalization/hi/taggers/tokenize_and_classify.py b/nemo_text_processing/text_normalization/hi/taggers/tokenize_and_classify.py index 3e1ded4b1..04124635e 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/tokenize_and_classify.py +++ b/nemo_text_processing/text_normalization/hi/taggers/tokenize_and_classify.py @@ -35,6 +35,8 @@ from nemo_text_processing.text_normalization.hi.taggers.money import MoneyFst from nemo_text_processing.text_normalization.hi.taggers.ordinal import OrdinalFst from nemo_text_processing.text_normalization.hi.taggers.punctuation import PunctuationFst +from nemo_text_processing.text_normalization.hi.taggers.roman import RomanFst +from nemo_text_processing.text_normalization.hi.taggers.serial import SerialFst from nemo_text_processing.text_normalization.hi.taggers.telephone import TelephoneFst from nemo_text_processing.text_normalization.hi.taggers.time import TimeFst from nemo_text_processing.text_normalization.hi.taggers.whitelist import WhiteListFst @@ -98,7 +100,12 @@ def __init__( ordinal = OrdinalFst(cardinal=cardinal, deterministic=deterministic) ordinal_graph = ordinal.fst - measure = MeasureFst(cardinal=cardinal, decimal=decimal, ordinal=ordinal, input_case=input_case) + serial = SerialFst(cardinal=cardinal, deterministic=deterministic) + serial_graph = serial.fst + + measure = MeasureFst( + cardinal=cardinal, decimal=decimal, ordinal=ordinal, serial=serial, input_case=input_case + ) measure_graph = measure.fst money = MoneyFst(cardinal=cardinal) @@ -111,6 +118,12 @@ def __init__( punctuation = PunctuationFst(deterministic=deterministic) punct_graph = punctuation.fst + word = WordFst(punctuation=punctuation, deterministic=deterministic) + word_graph = word.fst + + roman = RomanFst(deterministic=deterministic) + roman_graph = roman.fst + telephone = TelephoneFst() telephone_graph = telephone.fst @@ -121,7 +134,7 @@ def __init__( pynutil.add_weight(whitelist_graph, 1.01) | pynutil.add_weight(cardinal_graph, 1.1) | pynutil.add_weight(decimal_graph, 1.1) - | pynutil.add_weight(fraction_graph, 1.1) + | pynutil.add_weight(fraction_graph, 1.05) | pynutil.add_weight(date_graph, 1.1) | pynutil.add_weight(time_graph, 1.1) | pynutil.add_weight(measure_graph, 1.1) @@ -129,10 +142,10 @@ def __init__( | pynutil.add_weight(telephone_graph, 1.1) | pynutil.add_weight(ordinal_graph, 1.1) | pynutil.add_weight(electronic_graph, 1.1) + | pynutil.add_weight(serial_graph, 1.11) + | pynutil.add_weight(roman_graph, 1.1) ) - word_graph = WordFst(punctuation=punctuation, deterministic=deterministic).fst - punct = pynutil.insert("tokens { ") + pynutil.add_weight(punct_graph, weight=2.1) + pynutil.insert(" }") punct = pynini.closure( pynini.union( diff --git a/nemo_text_processing/text_normalization/hi/verbalizers/electronic.py b/nemo_text_processing/text_normalization/hi/verbalizers/electronic.py index 124e1c60b..29dcf81d9 100644 --- a/nemo_text_processing/text_normalization/hi/verbalizers/electronic.py +++ b/nemo_text_processing/text_normalization/hi/verbalizers/electronic.py @@ -16,6 +16,7 @@ from pynini.lib import pynutil from nemo_text_processing.text_normalization.hi.graph_utils import ( + NEMO_ALPHA, GraphFst, capitalized_input_graph, delete_space, @@ -27,13 +28,15 @@ class ElectronicFst(GraphFst): """ Finite state transducer for verbalizing electronic addresses. - Uses a phonetic-first approach with letter-by-letter fallback. + English words and letters are kept verbatim (Latin script); only digits and + symbols are read out in Hindi. Examples: - electronic { username: "kumar" domain: "gmail.com" } -> "कुमार एट जीमेल डॉट कॉम" - electronic { protocol: "https" domain: "google.com/" } -> "एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश गूगल डॉट कॉम फॉरवर्ड स्लैश" - electronic { path: "C:\\Users\\HP" } -> "सी कोलन बैकवर्ड स्लैश यूज़र्स बैकवर्ड स्लैश एच पी" - electronic { ip: "192.168.1.1" } -> "एक नौ दो डॉट एक छह आठ डॉट एक डॉट एक" + electronic { username: "kumar" domain: "gmail.com" } -> "kumar एट gmail डॉट com" + electronic { protocol: "https" domain: "google.com/" } -> "https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश google डॉट com फॉरवर्ड स्लैश" + electronic { path: "C:\\Users\\HP\\Desktop" } -> "C कोलन बैकवर्ड स्लैश Users बैकवर्ड स्लैश HP बैकवर्ड स्लैश Desktop" + electronic { domain: "192.168.1.1" } -> "एक नौ दो डॉट एक छह आठ डॉट एक डॉट एक" + electronic { fragment_id: "C₂H₄" } -> "सी दो एच चार" Args: deterministic: if True will provide a single transduction option, @@ -43,108 +46,73 @@ class ElectronicFst(GraphFst): def __init__(self, deterministic: bool = True): super().__init__(name="electronic", kind="verbalize", deterministic=deterministic) - # Load data files symbols_graph = pynini.string_file(get_abs_path("data/electronic/symbols.tsv")).optimize() - domain_graph = pynini.string_file(get_abs_path("data/electronic/domain.tsv")).optimize() - server_name_graph = pynini.string_file(get_abs_path("data/electronic/server_name.tsv")).optimize() - common_words_graph = pynini.string_file(get_abs_path("data/electronic/common_words.tsv")).optimize() - latin_to_hindi_graph = pynini.string_file(get_abs_path("data/address/letters.tsv")) - latin_to_hindi_graph = capitalized_input_graph(latin_to_hindi_graph).optimize() - # Digit mappings - use telephone number mappings for ASCII digits ascii_digit_graph = pynini.string_file(get_abs_path("data/telephone/number.tsv")).optimize() hindi_digit_graph = pynini.string_file(get_abs_path("data/numbers/digit.tsv")).optimize() hindi_zero_graph = pynini.string_file(get_abs_path("data/numbers/zero.tsv")).optimize() subscript_digit_graph = pynini.string_file(get_abs_path("data/electronic/subscript_digit.tsv")).optimize() digit_verbalization = ascii_digit_graph | hindi_digit_graph | hindi_zero_graph | subscript_digit_graph - # Combined phonetic word graph: server names + common words - phonetic_word = server_name_graph | common_words_graph - - # ============ CHARACTER VERBALIZATION ============ - # Single character to Hindi verbalization with space insertion - char_to_hindi = pynutil.add_weight(latin_to_hindi_graph, 1.0) | pynutil.add_weight( # Letter mapping - digit_verbalization, 1.0 - ) # Digit mapping - char_with_space = char_to_hindi + insert_space - - # ============ SYMBOL VERBALIZATION ============ - symbol_to_hindi = symbols_graph + insert_space - - # ============ DOMAIN VERBALIZATION ============ - # Domain extension verbalization (.com -> डॉट कॉम) - domain_ext_verbalization = pynini.cross(".", "डॉट ") + domain_graph + insert_space - - # ============ PROTOCOL VERBALIZATION ============ protocol_graph = pynini.string_file(get_abs_path("data/electronic/protocols.tsv")).optimize() - protocol_verbalization = protocol_graph + insert_space - # ============ FIELD EXTRACTION ============ - # Extract username field + single_digit = digit_verbalization + insert_space + single_symbol = symbols_graph + insert_space + + single_non_alpha = pynutil.add_weight(single_symbol, 1.0) | pynutil.add_weight(single_digit, 1.0) + + # A run of Latin letters is preserved verbatim; digits and symbols verbalize in Hindi. + alpha_run = pynini.closure(NEMO_ALPHA, 1) + insert_space + + # Chemical formulas are spelled out letter-by-letter (element symbols are + # abbreviations, not words), while digits and symbols verbalize in Hindi. + latin_to_hindi_graph = capitalized_input_graph( + pynini.string_file(get_abs_path("data/address/letters.tsv")) + ).optimize() + chem_char = (latin_to_hindi_graph + insert_space) | single_digit | single_symbol + chem_content = pynini.closure(chem_char, 1) + + def make_content(non_alpha_sep=None): + if non_alpha_sep is None: + non_alpha_sep = single_non_alpha + mandatory_sep = pynini.closure(non_alpha_sep, 1) + return ( + pynini.closure(non_alpha_sep, 0) + + pynini.closure(alpha_run + mandatory_sep, 0) + + pynini.closure(alpha_run, 0, 1) + + pynini.closure(non_alpha_sep, 0) + ) + delete_username_tag = pynutil.delete("username: \"") delete_domain_tag = pynutil.delete("domain: \"") delete_protocol_tag = pynutil.delete("protocol: \"") delete_path_tag = pynutil.delete("path: \"") + delete_fragment_id_tag = pynutil.delete("fragment_id: \"") delete_quote = pynutil.delete("\"") - # Username verbalization: letter-by-letter with symbol handling - username_content = pynini.closure( - pynutil.add_weight(phonetic_word + insert_space, 0.9) - | pynutil.add_weight(symbol_to_hindi, 1.0) - | pynutil.add_weight(char_with_space, 1.1), - 1, - ) - - username_graph = ( - delete_username_tag + username_content + delete_quote + delete_space + pynutil.insert("एट ") # @ symbol - ) - - # Domain verbalization - domain_content = pynini.closure( - pynutil.add_weight(phonetic_word + insert_space, 0.9) - | pynutil.add_weight(domain_ext_verbalization, 0.95) - | pynutil.add_weight(symbol_to_hindi, 1.0) - | pynutil.add_weight(char_with_space, 1.1), - 1, - ) - - domain_only_graph = delete_domain_tag + domain_content + delete_quote - - # Protocol verbalization - protocol_only_graph = delete_protocol_tag + protocol_verbalization + delete_quote + delete_space + general_content = make_content() - # Path verbalization (Windows/Unix file paths) - path_content = pynini.closure( - pynutil.add_weight(common_words_graph + insert_space, 0.9) - | pynutil.add_weight(symbol_to_hindi, 1.0) - | pynutil.add_weight(char_with_space, 1.1), - 1, - ) + username_graph = delete_username_tag + general_content + delete_quote + delete_space + pynutil.insert("एट ") + domain_only_graph = delete_domain_tag + general_content + delete_quote + protocol_only_graph = delete_protocol_tag + protocol_graph + insert_space + delete_quote + delete_space + path_graph = delete_path_tag + general_content + delete_quote - path_graph = delete_path_tag + path_content + delete_quote + chem_graph = delete_fragment_id_tag + chem_content + delete_quote - # IP address verbalization (digit by digit) - ip_char = pynutil.add_weight(symbols_graph + insert_space, 1.0) | pynutil.add_weight( - digit_verbalization + insert_space, 1.0 - ) + ip_char = single_symbol | single_digit ip_content = pynini.closure(ip_char, 1) - ip_graph = delete_domain_tag + ip_content + delete_quote - # ============ COMBINED GRAPH ============ - # Email: username + domain email_full = username_graph + domain_only_graph - - # URL with protocol: protocol + domain url_full = protocol_only_graph + domain_only_graph - # Combined final graph graph = ( pynutil.add_weight(url_full, 1.0) | pynutil.add_weight(email_full, 1.01) | pynutil.add_weight(path_graph, 1.02) | pynutil.add_weight(ip_graph, 1.03) | pynutil.add_weight(domain_only_graph, 1.04) + | pynutil.add_weight(chem_graph, 1.04) ) delete_tokens = self.delete_tokens(graph) diff --git a/nemo_text_processing/text_normalization/hi/verbalizers/fraction.py b/nemo_text_processing/text_normalization/hi/verbalizers/fraction.py index a07c41eae..66d944ea7 100644 --- a/nemo_text_processing/text_normalization/hi/verbalizers/fraction.py +++ b/nemo_text_processing/text_normalization/hi/verbalizers/fraction.py @@ -21,8 +21,8 @@ class FractionFst(GraphFst): """ Finite state transducer for verbalizing fraction - e.g. fraction { integer: "तेईस" numerator: "चार" denominator: "छः" }-> तेईस चार बटा छः - e.g. fraction { numerator: "चार" denominator: "छः" } -> चार बटा छः + e.g. fraction { integer: "तेईस" numerator: "चार" denominator: "छह" }-> तेईस और चार बटा छह + e.g. fraction { numerator: "चार" denominator: "छह" } -> चार बटा छह Args: diff --git a/nemo_text_processing/text_normalization/hi/verbalizers/money.py b/nemo_text_processing/text_normalization/hi/verbalizers/money.py index 048140295..1e5da99e4 100644 --- a/nemo_text_processing/text_normalization/hi/verbalizers/money.py +++ b/nemo_text_processing/text_normalization/hi/verbalizers/money.py @@ -15,26 +15,16 @@ import pynini from pynini.lib import pynutil -major_minor_currencies = { - "रुपए": "पैसे", - "पाउंड": "पेंस", - "वॉन": "जिओन", - "डॉलर": "सेंट", - "लीरा": "कुरस", - "टका": "पैसे", - "येन": "सेन", - "नाइरा": "कोबो", - "यूरो": "सेंट", -} from nemo_text_processing.text_normalization.hi.graph_utils import NEMO_NOT_QUOTE, NEMO_SPACE, GraphFst +from nemo_text_processing.text_normalization.hi.utils import get_abs_path, load_labels class MoneyFst(GraphFst): """ Finite state transducer for verbalizing money, e.g. - money { integer_part: "बारह" currency_maj: "रुपए" } -> बारह रुपए - money { integer_part: "बारह" currency_maj: "रुपए" fractional_part: "पचास" currency_min: "centiles" } -> बारह रुपए पचास पैसे - money { currency_maj: "रुपए" integer_part: "शून्य" fractional_part: "पचास" currency_min: "centiles" } -> पचास पैसे + money { currency_maj: "रुपए" integer_part: "बारह" } } -> बारह रुपए + money { currency_maj: "रुपए" integer_part: "बारह" fractional_part: "पचास" currency_min: "पैसे" } -> बारह रुपए पचास पैसे + money { currency_maj: "रुपए" integer_part: "शून्य" fractional_part: "पचास" currency_min: "पैसे" } -> पचास पैसे Args: cardinal: CardinalFst @@ -46,55 +36,67 @@ class MoneyFst(GraphFst): def __init__(self): super().__init__(name="money", kind="verbalize") - currency_major = pynutil.delete('currency_maj: "') + pynini.closure(NEMO_NOT_QUOTE, 1) + pynutil.delete('"') + sp = pynini.accep(NEMO_SPACE) + currency_major = pynutil.delete('currency_maj: "') + pynini.closure(NEMO_NOT_QUOTE, 1) + pynutil.delete('"') integer_part = pynutil.delete('integer_part: "') + pynini.closure(NEMO_NOT_QUOTE, 1) + pynutil.delete('"') - fractional_part = ( pynutil.delete('fractional_part: "') + pynini.closure(NEMO_NOT_QUOTE, 1) + pynutil.delete('"') ) - # Handles major denominations only - graph_major_only = integer_part + pynini.accep(NEMO_SPACE) + currency_major + currency_minor = pynutil.delete('currency_min: "') + pynini.closure(NEMO_NOT_QUOTE, 1) + pynutil.delete('"') + + graph_major_only = integer_part + sp + currency_major + + all_major_names = [maj for maj, _ in load_labels(get_abs_path("data/money/major_minor_currencies.tsv"))] - # Handles both major and minor denominations major_minor_graphs = [] + minor_only_graphs = [] + + for major in all_major_names: + graph_major_slot = pynutil.delete('currency_maj: "') + pynutil.delete(major) + pynutil.delete('"') + + major_minor_graphs.append( + graph_major_slot + + sp + + integer_part + + pynutil.insert(NEMO_SPACE) + + pynutil.insert(major) + + sp + + fractional_part + + sp + + currency_minor + ) - # Handles minor denominations only - minor_graphs = [] - - # Logic for handling minor denominations - for major, minor in major_minor_currencies.items(): - graph_major = pynutil.delete('currency_maj: "') + pynini.accep(major) + pynutil.delete('"') - graph_minor = pynutil.delete('currency_min: "') + pynini.cross("centiles", minor) + pynutil.delete('"') - graph_major_minor_partial = ( - integer_part - + pynini.accep(NEMO_SPACE) - + graph_major - + pynini.accep(NEMO_SPACE) + minor_only_graphs.append( + graph_major_slot + + sp + + pynutil.delete('integer_part: "शून्य"') + + sp + fractional_part - + pynini.accep(NEMO_SPACE) - + graph_minor + + sp + + currency_minor ) - major_minor_graphs.append(graph_major_minor_partial) - graph_minor_partial = ( - pynutil.delete('integer_part: "शून्य"') - + pynutil.delete(NEMO_SPACE) - + pynutil.delete('currency_maj: "') + graph_major_minor = pynini.union(*major_minor_graphs) + graph_minor_only = pynini.union(*minor_only_graphs) + + decimal_graphs = [] + for major in all_major_names: + decimal_graphs.append( + pynutil.delete('currency_maj: "') + pynutil.delete(major) + pynutil.delete('"') - + pynutil.delete(NEMO_SPACE) + + sp + + integer_part + + sp + + pynutil.insert(" दशमलव ") + fractional_part - + pynini.accep(NEMO_SPACE) - + graph_minor + + pynutil.insert(NEMO_SPACE) + + pynutil.insert(major) ) - minor_graphs.append(graph_minor_partial) - - graph_major_minor = pynini.union(*major_minor_graphs) - graph_minor_only = pynini.union(*minor_graphs) + graph_decimal_money = pynini.union(*decimal_graphs) - graph = graph_major_only | graph_major_minor | pynutil.add_weight(graph_minor_only, -0.1) + graph = graph_major_only | graph_major_minor | pynutil.add_weight(graph_minor_only, -0.1) | graph_decimal_money - delete_tokens = self.delete_tokens(graph) - self.fst = delete_tokens.optimize() + self.fst = self.delete_tokens(graph).optimize() diff --git a/nemo_text_processing/text_normalization/hi/verbalizers/roman.py b/nemo_text_processing/text_normalization/hi/verbalizers/roman.py new file mode 100644 index 000000000..c28084a77 --- /dev/null +++ b/nemo_text_processing/text_normalization/hi/verbalizers/roman.py @@ -0,0 +1,98 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.text_normalization.hi.graph_utils import ( + NEMO_NOT_QUOTE, + GraphFst, + delete_zero_or_one_space, + insert_space, +) +from nemo_text_processing.text_normalization.hi.utils import get_abs_path + + +class RomanFst(GraphFst): + """ + Finite state transducer for verbalizing Roman numerals in Hindi. + roman { preserve_order: true key_cardinal: "भास्कर" integer: "II" } -> भास्कर दो + roman { preserve_order: true key_cardinal: "कक्षा" integer: "XII" } -> कक्षा बारह + roman { preserve_order: true integer: "XII" default_ordinal: "बारहवीं" key_cardinal: "कक्षा" } -> बारहवीं कक्षा + roman { preserve_order: true integer: "IV" default_ordinal: "चौथी" key_cardinal: "कक्षा" } -> चौथी कक्षा + + Args: + deterministic: if True will provide a single transduction option, + for False multiple options (used for audio-based normalization) + """ + + def __init__(self, deterministic: bool = True): + super().__init__(name="roman", kind="verbalize", deterministic=deterministic) + + roman_to_spoken = pynini.string_file(get_abs_path("data/roman/roman_to_spoken.tsv")).optimize() + + key_cardinal = ( + pynutil.delete('key_cardinal: "') + pynini.closure(NEMO_NOT_QUOTE, 1) + pynutil.delete('"') + ).optimize() + + integer = (pynutil.delete('integer: "') + roman_to_spoken + pynutil.delete('"')).optimize() + + default_ordinal = ( + pynutil.delete('default_ordinal: "') + pynini.closure(NEMO_NOT_QUOTE, 1) + pynutil.delete('"') + ).optimize() + + ignore_integer = ( + pynutil.delete('integer: "') + pynutil.delete(pynini.closure(NEMO_NOT_QUOTE, 1)) + pynutil.delete('"') + ).optimize() + + drop_preserve_order = pynini.closure( + delete_zero_or_one_space + + pynutil.delete("preserve_order:") + + delete_zero_or_one_space + + pynutil.delete("true") + + delete_zero_or_one_space, + 0, + 1, + ).optimize() + + key_first = ( + drop_preserve_order + + key_cardinal + + delete_zero_or_one_space + + insert_space + + integer + + drop_preserve_order + ).optimize() + + numeral_first = ( + drop_preserve_order + + integer + + delete_zero_or_one_space + + insert_space + + key_cardinal + + drop_preserve_order + ).optimize() + + glued_ordinal = ( + drop_preserve_order + + ignore_integer + + delete_zero_or_one_space + + default_ordinal + + pynini.closure(delete_zero_or_one_space + insert_space + key_cardinal, 0, 1) + + drop_preserve_order + ).optimize() + + graph = pynini.union(key_first, numeral_first, glued_ordinal).optimize() + + self.fst = self.delete_tokens(graph).optimize() diff --git a/nemo_text_processing/text_normalization/hi/verbalizers/verbalize.py b/nemo_text_processing/text_normalization/hi/verbalizers/verbalize.py index e0fb8d8b5..bd6ca4b5b 100644 --- a/nemo_text_processing/text_normalization/hi/verbalizers/verbalize.py +++ b/nemo_text_processing/text_normalization/hi/verbalizers/verbalize.py @@ -21,6 +21,7 @@ from nemo_text_processing.text_normalization.hi.verbalizers.measure import MeasureFst from nemo_text_processing.text_normalization.hi.verbalizers.money import MoneyFst from nemo_text_processing.text_normalization.hi.verbalizers.ordinal import OrdinalFst +from nemo_text_processing.text_normalization.hi.verbalizers.roman import RomanFst from nemo_text_processing.text_normalization.hi.verbalizers.telephone import TelephoneFst from nemo_text_processing.text_normalization.hi.verbalizers.time import TimeFst from nemo_text_processing.text_normalization.hi.verbalizers.whitelist import WhiteListFst @@ -70,6 +71,9 @@ def __init__(self, deterministic: bool = True): electronic = ElectronicFst(deterministic=deterministic) electronic_graph = electronic.fst + roman = RomanFst(deterministic=deterministic) + roman_graph = roman.fst + whitelist_graph = WhiteListFst(deterministic=deterministic).fst graph = ( @@ -84,6 +88,7 @@ def __init__(self, deterministic: bool = True): | whitelist_graph | telephone_graph | electronic_graph + | roman_graph ) self.fst = graph diff --git a/nemo_text_processing/text_normalization/ja/data/address/__init__.py b/nemo_text_processing/text_normalization/ja/data/address/__init__.py new file mode 100644 index 000000000..9e3fb699d --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/address/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/text_normalization/ja/data/address/postal.tsv b/nemo_text_processing/text_normalization/ja/data/address/postal.tsv new file mode 100644 index 000000000..2ed52b275 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/address/postal.tsv @@ -0,0 +1 @@ +〒 郵便番号 diff --git a/nemo_text_processing/text_normalization/ja/data/address/room_suffix.tsv b/nemo_text_processing/text_normalization/ja/data/address/room_suffix.tsv new file mode 100644 index 000000000..6d259df0d --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/address/room_suffix.tsv @@ -0,0 +1 @@ +号室 diff --git a/nemo_text_processing/text_normalization/ja/data/address/separator.tsv b/nemo_text_processing/text_normalization/ja/data/address/separator.tsv new file mode 100644 index 000000000..028b8020d --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/address/separator.tsv @@ -0,0 +1,3 @@ +- の +- の +ー の diff --git a/nemo_text_processing/text_normalization/ja/data/date/day.tsv b/nemo_text_processing/text_normalization/ja/data/date/day.tsv index 09258cb4c..cd6dbe49a 100644 --- a/nemo_text_processing/text_normalization/ja/data/date/day.tsv +++ b/nemo_text_processing/text_normalization/ja/data/date/day.tsv @@ -28,4 +28,4 @@ 28 二十八 29 二十九 30 三十 -31 三十一 \ No newline at end of file +31 三十一 diff --git a/nemo_text_processing/text_normalization/ja/data/date/era.tsv b/nemo_text_processing/text_normalization/ja/data/date/era.tsv index b932b7fb3..85e56bce0 100644 --- a/nemo_text_processing/text_normalization/ja/data/date/era.tsv +++ b/nemo_text_processing/text_normalization/ja/data/date/era.tsv @@ -9,4 +9,4 @@ グレゴリオ暦 紀元前 紀元 -紀元後 \ No newline at end of file +紀元後 diff --git a/nemo_text_processing/text_normalization/ja/data/date/era_abbrev.tsv b/nemo_text_processing/text_normalization/ja/data/date/era_abbrev.tsv index 1c95675c2..996934a15 100644 --- a/nemo_text_processing/text_normalization/ja/data/date/era_abbrev.tsv +++ b/nemo_text_processing/text_normalization/ja/data/date/era_abbrev.tsv @@ -2,4 +2,4 @@ R. 令和 H. 平成 S. 昭和 T. 大正 -M. 明治 \ No newline at end of file +M. 明治 diff --git a/nemo_text_processing/text_normalization/ja/data/date/month.tsv b/nemo_text_processing/text_normalization/ja/data/date/month.tsv index f992b4d28..33896fe82 100644 --- a/nemo_text_processing/text_normalization/ja/data/date/month.tsv +++ b/nemo_text_processing/text_normalization/ja/data/date/month.tsv @@ -9,4 +9,4 @@ 9 九 10 十 11 十一 -12 十二 \ No newline at end of file +12 十二 diff --git a/nemo_text_processing/text_normalization/ja/data/date/suffix.tsv b/nemo_text_processing/text_normalization/ja/data/date/suffix.tsv new file mode 100644 index 000000000..a1064b515 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/date/suffix.tsv @@ -0,0 +1,8 @@ +year 年 +month 月 +day 日 +century 世紀 +decade 年代 +early_ten_days 上旬 +middle_ten_days 中旬 +late_ten_days 下旬 diff --git a/nemo_text_processing/text_normalization/ja/data/date/week.tsv b/nemo_text_processing/text_normalization/ja/data/date/week.tsv index 41556b4db..af2e05c5c 100644 --- a/nemo_text_processing/text_normalization/ja/data/date/week.tsv +++ b/nemo_text_processing/text_normalization/ja/data/date/week.tsv @@ -12,4 +12,4 @@ 木曜日 金曜日 土曜日 -日曜日 \ No newline at end of file +日曜日 diff --git a/nemo_text_processing/text_normalization/ja/data/electronic/__init__.py b/nemo_text_processing/text_normalization/ja/data/electronic/__init__.py new file mode 100644 index 000000000..9e3fb699d --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/electronic/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/text_normalization/ja/data/electronic/card_cues.tsv b/nemo_text_processing/text_normalization/ja/data/electronic/card_cues.tsv new file mode 100644 index 000000000..769dee97d --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/electronic/card_cues.tsv @@ -0,0 +1,2 @@ +カード末尾 +カード番号 diff --git a/nemo_text_processing/text_normalization/ja/data/electronic/card_digit_count_prefix.tsv b/nemo_text_processing/text_normalization/ja/data/electronic/card_digit_count_prefix.tsv new file mode 100644 index 000000000..4a52fb771 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/electronic/card_digit_count_prefix.tsv @@ -0,0 +1 @@ +カード下 diff --git a/nemo_text_processing/text_normalization/ja/data/electronic/card_digit_count_suffix.tsv b/nemo_text_processing/text_normalization/ja/data/electronic/card_digit_count_suffix.tsv new file mode 100644 index 000000000..d4fb9db66 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/electronic/card_digit_count_suffix.tsv @@ -0,0 +1 @@ +桁 diff --git a/nemo_text_processing/text_normalization/ja/data/electronic/file_extensions.tsv b/nemo_text_processing/text_normalization/ja/data/electronic/file_extensions.tsv new file mode 100644 index 000000000..96c957484 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/electronic/file_extensions.tsv @@ -0,0 +1,6 @@ +.jpg ドット jpg +.JPG ドット JPG +.png ドット png +.PNG ドット PNG +.pdf ドット pdf +.PDF ドット PDF diff --git a/nemo_text_processing/text_normalization/ja/data/electronic/protocol.tsv b/nemo_text_processing/text_normalization/ja/data/electronic/protocol.tsv new file mode 100644 index 000000000..d1cfd2593 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/electronic/protocol.tsv @@ -0,0 +1,2 @@ +http +https diff --git a/nemo_text_processing/text_normalization/ja/data/electronic/special_digit_runs.tsv b/nemo_text_processing/text_normalization/ja/data/electronic/special_digit_runs.tsv new file mode 100644 index 000000000..8609135f3 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/electronic/special_digit_runs.tsv @@ -0,0 +1 @@ +01 ゼロイチ diff --git a/nemo_text_processing/text_normalization/ja/data/electronic/symbol.tsv b/nemo_text_processing/text_normalization/ja/data/electronic/symbol.tsv new file mode 100644 index 000000000..568588980 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/electronic/symbol.tsv @@ -0,0 +1,5 @@ +@ アット +. ドット +- ハイフン +/ スラッシュ +: コロン diff --git a/nemo_text_processing/text_normalization/ja/data/fraction/__init__.py b/nemo_text_processing/text_normalization/ja/data/fraction/__init__.py new file mode 100644 index 000000000..9e3fb699d --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/fraction/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/text_normalization/ja/data/fraction/marker.tsv b/nemo_text_processing/text_normalization/ja/data/fraction/marker.tsv new file mode 100644 index 000000000..23fe7022a --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/fraction/marker.tsv @@ -0,0 +1,4 @@ +fraction 分の +root_written √ +root_spoken ルート +mixed と diff --git a/nemo_text_processing/text_normalization/ja/data/latin/__init__.py b/nemo_text_processing/text_normalization/ja/data/latin/__init__.py new file mode 100644 index 000000000..9e3fb699d --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/latin/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/text_normalization/ja/data/latin/letters.tsv b/nemo_text_processing/text_normalization/ja/data/latin/letters.tsv new file mode 100644 index 000000000..c28af1201 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/latin/letters.tsv @@ -0,0 +1,26 @@ +A エー +B ビー +C シー +D ディー +E イー +F エフ +G ジー +H エイチ +I アイ +J ジェー +K ケー +L エル +M エム +N エヌ +O オー +P ピー +Q キュー +R アール +S エス +T ティー +U ユー +V ブイ +W ダブリュー +X エックス +Y ワイ +Z ゼット diff --git a/nemo_text_processing/text_normalization/ja/data/measure/__init__.py b/nemo_text_processing/text_normalization/ja/data/measure/__init__.py new file mode 100644 index 000000000..9e3fb699d --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/measure/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/text_normalization/ja/data/measure/per_marker.tsv b/nemo_text_processing/text_normalization/ja/data/measure/per_marker.tsv new file mode 100644 index 000000000..133297769 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/measure/per_marker.tsv @@ -0,0 +1 @@ +毎 diff --git a/nemo_text_processing/text_normalization/ja/data/measure/per_unit.tsv b/nemo_text_processing/text_normalization/ja/data/measure/per_unit.tsv new file mode 100644 index 000000000..ed878a1bf --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/measure/per_unit.tsv @@ -0,0 +1,33 @@ +h 時 +h 時 +hour 時 +hours 時 +s 秒 +s 秒 +sec 秒 +second 秒 +seconds 秒 +min 分 +min 分 +minute 分 +minutes 分 +kg キロ +kg キロ +g グラム +g グラム +m メートル +m メートル +km キロ +km キロ +cm センチメートル +cm センチメートル +mm ミリメートル +mm ミリメートル +L リットル +L リットル +l リットル +l リットル +ml ミリリットル +mL ミリリットル +ml ミリリットル +mL ミリリットル diff --git a/nemo_text_processing/text_normalization/ja/data/measure/rate_numerator.tsv b/nemo_text_processing/text_normalization/ja/data/measure/rate_numerator.tsv new file mode 100644 index 000000000..126b4a3e1 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/measure/rate_numerator.tsv @@ -0,0 +1,4 @@ +km キロ +km キロ +m メートル +m メートル diff --git a/nemo_text_processing/text_normalization/ja/data/measure/speed.tsv b/nemo_text_processing/text_normalization/ja/data/measure/speed.tsv new file mode 100644 index 000000000..275b2c14d --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/measure/speed.tsv @@ -0,0 +1,2 @@ +キロ 時 時速 +メートル 秒 秒速 diff --git a/nemo_text_processing/text_normalization/ja/data/measure/unit.tsv b/nemo_text_processing/text_normalization/ja/data/measure/unit.tsv new file mode 100644 index 000000000..0573068f8 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/measure/unit.tsv @@ -0,0 +1,79 @@ +kg キロ +kg キロ +キロ キロ +キログラム キログラム +g グラム +g グラム +グラム グラム +mg ミリグラム +mg ミリグラム +ミリグラム ミリグラム +μg マイクログラム +µg マイクログラム +マイクログラム マイクログラム +km キロ +km キロ +キロメートル キロ +m メートル +m メートル +メートル メートル +cm センチ +cm センチ +センチ センチ +センチメートル センチメートル +mm ミリメートル +mm ミリメートル +ミリメートル ミリメートル +m2 平方メートル +m² 平方メートル +m2 平方メートル +m² 平方メートル +平方メートル 平方メートル +cm2 平方センチメートル +cm² 平方センチメートル +平方センチメートル 平方センチメートル +L リットル +l リットル +L リットル +l リットル +リットル リットル +mL ミリリットル +ml ミリリットル +mL ミリリットル +ml ミリリットル +ミリリットル ミリリットル +% パーセント +% パーセント +パーセント パーセント +° 度 +º 度 +度 度 +°C 度 +℃ 度 +ºC 度 +°F 度エフ +℉ 度エフ +W ワット +w ワット +ワット ワット +kW キロワット +kw キロワット +キロワット キロワット +V ボルト +v ボルト +ボルト ボルト +Hz ヘルツ +hz ヘルツ +ヘルツ ヘルツ +kHz キロヘルツ +khz キロヘルツ +MHz メガヘルツ +mhz メガヘルツ +GHz ギガヘルツ +ghz ギガヘルツ +GB ギガバイト +gb ギガバイト +ギガバイト ギガバイト +MB メガバイト +mb メガバイト +メガバイト メガバイト diff --git a/nemo_text_processing/text_normalization/ja/data/money/__init__.py b/nemo_text_processing/text_normalization/ja/data/money/__init__.py new file mode 100644 index 000000000..4fc25d0d3 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/money/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/text_normalization/ja/data/money/currency_major.tsv b/nemo_text_processing/text_normalization/ja/data/money/currency_major.tsv new file mode 100644 index 000000000..c5f2efa84 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/money/currency_major.tsv @@ -0,0 +1,27 @@ +¥ 円 +¥ 円 +JPY 円 +jpy 円 +円 円 +$ ドル +US$ ドル +USD ドル +usd ドル +ドル ドル +米ドル ドル +€ ユーロ +EUR ユーロ +eur ユーロ +ユーロ ユーロ +£ ポンド +GBP ポンド +gbp ポンド +ポンド ポンド +₩ ウォン +KRW ウォン +krw ウォン +ウォン ウォン +CNY 元 +cny 元 +元 元 +人民元 人民元 diff --git a/nemo_text_processing/text_normalization/ja/data/money/currency_minor.tsv b/nemo_text_processing/text_normalization/ja/data/money/currency_minor.tsv new file mode 100644 index 000000000..3709c9572 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/money/currency_minor.tsv @@ -0,0 +1,2 @@ +銭 銭 +セント セント diff --git a/nemo_text_processing/text_normalization/ja/data/money/currency_minor_by_major.tsv b/nemo_text_processing/text_normalization/ja/data/money/currency_minor_by_major.tsv new file mode 100644 index 000000000..bdb9c4d07 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/money/currency_minor_by_major.tsv @@ -0,0 +1,15 @@ +¥ 銭 +¥ 銭 +JPY 銭 +jpy 銭 +円 銭 +$ セント +US$ セント +USD セント +usd セント +ドル セント +米ドル セント +€ セント +EUR セント +eur セント +ユーロ セント diff --git a/nemo_text_processing/text_normalization/ja/data/money/currency_prefix.tsv b/nemo_text_processing/text_normalization/ja/data/money/currency_prefix.tsv new file mode 100644 index 000000000..c959bf0ca --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/money/currency_prefix.tsv @@ -0,0 +1,19 @@ +¥ 円 +¥ 円 +JPY 円 +jpy 円 +$ ドル +US$ ドル +USD ドル +usd ドル +€ ユーロ +EUR ユーロ +eur ユーロ +£ ポンド +GBP ポンド +gbp ポンド +₩ ウォン +KRW ウォン +krw ウォン +CNY 元 +cny 元 diff --git a/nemo_text_processing/text_normalization/ja/data/money/quantity.tsv b/nemo_text_processing/text_normalization/ja/data/money/quantity.tsv new file mode 100644 index 000000000..bfbb13743 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/money/quantity.tsv @@ -0,0 +1,3 @@ +万 +億 +兆 diff --git a/nemo_text_processing/text_normalization/ja/data/numbers/decimal_point.tsv b/nemo_text_processing/text_normalization/ja/data/numbers/decimal_point.tsv new file mode 100644 index 000000000..ae902786c --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/numbers/decimal_point.tsv @@ -0,0 +1 @@ +. 点 diff --git a/nemo_text_processing/text_normalization/ja/data/numbers/digit.tsv b/nemo_text_processing/text_normalization/ja/data/numbers/digit.tsv index ede6b97b7..3a578d358 100644 --- a/nemo_text_processing/text_normalization/ja/data/numbers/digit.tsv +++ b/nemo_text_processing/text_normalization/ja/data/numbers/digit.tsv @@ -6,4 +6,4 @@ 6 六 7 七 8 八 -9 九 \ No newline at end of file +9 九 diff --git a/nemo_text_processing/text_normalization/ja/data/numbers/magnitude.tsv b/nemo_text_processing/text_normalization/ja/data/numbers/magnitude.tsv new file mode 100644 index 000000000..e5ecf396b --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/numbers/magnitude.tsv @@ -0,0 +1,4 @@ +hundred 百 +thousand 千 +ten_thousand 万 +hundred_million 億 diff --git a/nemo_text_processing/text_normalization/ja/data/numbers/sign.tsv b/nemo_text_processing/text_normalization/ja/data/numbers/sign.tsv new file mode 100644 index 000000000..76f8308b7 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/numbers/sign.tsv @@ -0,0 +1,2 @@ +- マイナス +マイナス マイナス diff --git a/nemo_text_processing/text_normalization/ja/data/numbers/teen.tsv b/nemo_text_processing/text_normalization/ja/data/numbers/teen.tsv index 1585fe836..52dc01917 100644 --- a/nemo_text_processing/text_normalization/ja/data/numbers/teen.tsv +++ b/nemo_text_processing/text_normalization/ja/data/numbers/teen.tsv @@ -7,4 +7,4 @@ 16 十六 17 十七 18 十八 -19 十九 \ No newline at end of file +19 十九 diff --git a/nemo_text_processing/text_normalization/ja/data/numbers/ties.tsv b/nemo_text_processing/text_normalization/ja/data/numbers/ties.tsv index 562e63265..2a73c0399 100644 --- a/nemo_text_processing/text_normalization/ja/data/numbers/ties.tsv +++ b/nemo_text_processing/text_normalization/ja/data/numbers/ties.tsv @@ -5,4 +5,4 @@ 6 六十 7 七十 8 八十 -9 九十 \ No newline at end of file +9 九十 diff --git a/nemo_text_processing/text_normalization/ja/data/numbers/zero.tsv b/nemo_text_processing/text_normalization/ja/data/numbers/zero.tsv index 7fab21718..ac94cc719 100644 --- a/nemo_text_processing/text_normalization/ja/data/numbers/zero.tsv +++ b/nemo_text_processing/text_normalization/ja/data/numbers/zero.tsv @@ -1 +1 @@ -0 零 \ No newline at end of file +0 ゼロ diff --git a/nemo_text_processing/text_normalization/ja/data/numbers/zero_decimal.tsv b/nemo_text_processing/text_normalization/ja/data/numbers/zero_decimal.tsv new file mode 100644 index 000000000..d6b9cece5 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/numbers/zero_decimal.tsv @@ -0,0 +1 @@ +0 零 diff --git a/nemo_text_processing/text_normalization/ja/data/numbers/zero_maru.tsv b/nemo_text_processing/text_normalization/ja/data/numbers/zero_maru.tsv new file mode 100644 index 000000000..67a5bfdbb --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/numbers/zero_maru.tsv @@ -0,0 +1 @@ +0 〇 diff --git a/nemo_text_processing/text_normalization/ja/data/ordinal/__init__.py b/nemo_text_processing/text_normalization/ja/data/ordinal/__init__.py new file mode 100644 index 000000000..9e3fb699d --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/ordinal/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/text_normalization/ja/data/ordinal/marker.tsv b/nemo_text_processing/text_normalization/ja/data/ordinal/marker.tsv new file mode 100644 index 000000000..fed01aeb8 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/ordinal/marker.tsv @@ -0,0 +1,2 @@ +prefix 第 +suffix 番目 diff --git a/nemo_text_processing/text_normalization/ja/data/post_processing/__init__.py b/nemo_text_processing/text_normalization/ja/data/post_processing/__init__.py new file mode 100644 index 000000000..9e3fb699d --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/post_processing/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/text_normalization/ja/data/post_processing/sentence_suffix.tsv b/nemo_text_processing/text_normalization/ja/data/post_processing/sentence_suffix.tsv new file mode 100644 index 000000000..ac7eb4c33 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/post_processing/sentence_suffix.tsv @@ -0,0 +1 @@ +です diff --git a/nemo_text_processing/text_normalization/ja/data/punctuation/__init__.py b/nemo_text_processing/text_normalization/ja/data/punctuation/__init__.py new file mode 100644 index 000000000..9e3fb699d --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/punctuation/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/text_normalization/ja/data/punctuation/extra.tsv b/nemo_text_processing/text_normalization/ja/data/punctuation/extra.tsv new file mode 100644 index 000000000..3cbc272d1 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/punctuation/extra.tsv @@ -0,0 +1,14 @@ +< +> +` +々 +ゝ +ゞ +ヽ +ヾ +〱 +〲 +〳 +〴 +〵 +〆 diff --git a/nemo_text_processing/text_normalization/ja/data/punctuation/range.tsv b/nemo_text_processing/text_normalization/ja/data/punctuation/range.tsv new file mode 100644 index 000000000..f83d02ca2 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/punctuation/range.tsv @@ -0,0 +1 @@ +〜 から diff --git a/nemo_text_processing/text_normalization/ja/data/range/__init__.py b/nemo_text_processing/text_normalization/ja/data/range/__init__.py new file mode 100644 index 000000000..9e3fb699d --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/range/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/text_normalization/ja/data/range/operator.tsv b/nemo_text_processing/text_normalization/ja/data/range/operator.tsv new file mode 100644 index 000000000..d412189b7 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/range/operator.tsv @@ -0,0 +1,2 @@ +x かける +× かける diff --git a/nemo_text_processing/text_normalization/ja/data/range/separator.tsv b/nemo_text_processing/text_normalization/ja/data/range/separator.tsv new file mode 100644 index 000000000..e774bb177 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/range/separator.tsv @@ -0,0 +1,2 @@ +- から +〜 から diff --git a/nemo_text_processing/text_normalization/ja/data/range/suffix.tsv b/nemo_text_processing/text_normalization/ja/data/range/suffix.tsv new file mode 100644 index 000000000..fd2a91bc5 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/range/suffix.tsv @@ -0,0 +1,2 @@ +人 +歳 diff --git a/nemo_text_processing/text_normalization/ja/data/range/year_suffix.tsv b/nemo_text_processing/text_normalization/ja/data/range/year_suffix.tsv new file mode 100644 index 000000000..76d279961 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/range/year_suffix.tsv @@ -0,0 +1 @@ +s 年代 diff --git a/nemo_text_processing/text_normalization/ja/data/roman/__init__.py b/nemo_text_processing/text_normalization/ja/data/roman/__init__.py new file mode 100644 index 000000000..9e3fb699d --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/roman/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/text_normalization/ja/data/roman/japanese_prefix.tsv b/nemo_text_processing/text_normalization/ja/data/roman/japanese_prefix.tsv new file mode 100644 index 000000000..4c34f1c09 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/roman/japanese_prefix.tsv @@ -0,0 +1 @@ +第 diff --git a/nemo_text_processing/text_normalization/ja/data/roman/japanese_suffix.tsv b/nemo_text_processing/text_normalization/ja/data/roman/japanese_suffix.tsv new file mode 100644 index 000000000..fa74f5ad8 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/roman/japanese_suffix.tsv @@ -0,0 +1,4 @@ +章 +条 +巻 +回 diff --git a/nemo_text_processing/text_normalization/ja/data/roman/key_cardinal.tsv b/nemo_text_processing/text_normalization/ja/data/roman/key_cardinal.tsv new file mode 100644 index 000000000..6b8f72d6d --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/roman/key_cardinal.tsv @@ -0,0 +1,14 @@ +Chapter +chapter +Part +part +Century +century +Class +class +Article +article +Section +section +Paragraph +paragraph diff --git a/nemo_text_processing/text_normalization/ja/data/roman/roman_numerals.tsv b/nemo_text_processing/text_normalization/ja/data/roman/roman_numerals.tsv new file mode 100644 index 000000000..f443fd3a9 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/roman/roman_numerals.tsv @@ -0,0 +1,13 @@ +I 1 +V 5 +X 10 +L 50 +C 100 +D 500 +M 1000 +IV 4 +IX 9 +XL 40 +XC 90 +CD 400 +CM 900 diff --git a/nemo_text_processing/text_normalization/ja/data/serial/__init__.py b/nemo_text_processing/text_normalization/ja/data/serial/__init__.py new file mode 100644 index 000000000..9e3fb699d --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/serial/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/text_normalization/ja/data/serial/delimiter.tsv b/nemo_text_processing/text_normalization/ja/data/serial/delimiter.tsv new file mode 100644 index 000000000..4a6fc82ef --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/serial/delimiter.tsv @@ -0,0 +1,2 @@ +- ハイフン +/ スラッシュ diff --git a/nemo_text_processing/text_normalization/ja/data/serial/model_cues.tsv b/nemo_text_processing/text_normalization/ja/data/serial/model_cues.tsv new file mode 100644 index 000000000..c3d4f5e18 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/serial/model_cues.tsv @@ -0,0 +1 @@ +型番 diff --git a/nemo_text_processing/text_normalization/ja/data/serial/words.tsv b/nemo_text_processing/text_normalization/ja/data/serial/words.tsv new file mode 100644 index 000000000..87725d259 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/serial/words.tsv @@ -0,0 +1,3 @@ +covid コビッド +COVID コビッド +Covid コビッド diff --git a/nemo_text_processing/text_normalization/ja/data/symbol.tsv b/nemo_text_processing/text_normalization/ja/data/symbol.tsv index 67ad785f0..c1dcfe15e 100644 --- a/nemo_text_processing/text_normalization/ja/data/symbol.tsv +++ b/nemo_text_processing/text_normalization/ja/data/symbol.tsv @@ -1,23 +1,23 @@ -& アンド -# ハッシュタグ -@ アット -§ セクション -™ トレードマーク -® 登録商標マーク -© 著作権 -_ アンダースコア -% パーセント -* 星印 -+ プラス -/ スラッシュ -= エコール -^ 曲折アクセント記号 -| 縦棒 -~ ティルダ -$ ドール -£ ポンド -€ ユーロ -₩ ウォン -¥ 円 -° 度 -º 度 \ No newline at end of file +& アンド +# ハッシュタグ +@ アット +§ セクション +™ トレードマーク +® 登録商標マーク +© 著作権 +_ アンダースコア +% パーセント +* 星印 ++ プラス +/ スラッシュ += エコール +^ 曲折アクセント記号 +| 縦棒 +~ ティルダ +$ ドール +£ ポンド +€ ユーロ +₩ ウォン +¥ 円 +° 度 +º 度 diff --git a/nemo_text_processing/text_normalization/ja/data/telephone/__init__.py b/nemo_text_processing/text_normalization/ja/data/telephone/__init__.py new file mode 100644 index 000000000..9e3fb699d --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/telephone/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/text_normalization/ja/data/telephone/country_code_prefix.tsv b/nemo_text_processing/text_normalization/ja/data/telephone/country_code_prefix.tsv new file mode 100644 index 000000000..209f307d4 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/telephone/country_code_prefix.tsv @@ -0,0 +1 @@ +プラス diff --git a/nemo_text_processing/text_normalization/ja/data/telephone/extension.tsv b/nemo_text_processing/text_normalization/ja/data/telephone/extension.tsv new file mode 100644 index 000000000..6f19d8b99 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/telephone/extension.tsv @@ -0,0 +1,2 @@ +内線番号 内線 +内線 内線 diff --git a/nemo_text_processing/text_normalization/ja/data/telephone/group_separator.tsv b/nemo_text_processing/text_normalization/ja/data/telephone/group_separator.tsv new file mode 100644 index 000000000..f43a2f5e6 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/telephone/group_separator.tsv @@ -0,0 +1 @@ +、 diff --git a/nemo_text_processing/text_normalization/ja/data/time/division.tsv b/nemo_text_processing/text_normalization/ja/data/time/division.tsv index f6beedce8..59b40ce7c 100644 --- a/nemo_text_processing/text_normalization/ja/data/time/division.tsv +++ b/nemo_text_processing/text_normalization/ja/data/time/division.tsv @@ -20,4 +20,4 @@ 翌日 未明 正午 -真夜中の \ No newline at end of file +真夜中の diff --git a/nemo_text_processing/text_normalization/ja/data/time/hour.tsv b/nemo_text_processing/text_normalization/ja/data/time/hour.tsv index 1d5c08425..14ef0cf3c 100644 --- a/nemo_text_processing/text_normalization/ja/data/time/hour.tsv +++ b/nemo_text_processing/text_normalization/ja/data/time/hour.tsv @@ -21,4 +21,4 @@ 21 二十一 22 二十二 23 二十三 -24 二十四 \ No newline at end of file +24 二十四 diff --git a/nemo_text_processing/text_normalization/ja/data/time/minute.tsv b/nemo_text_processing/text_normalization/ja/data/time/minute.tsv deleted file mode 100644 index 5e8276a00..000000000 --- a/nemo_text_processing/text_normalization/ja/data/time/minute.tsv +++ /dev/null @@ -1,60 +0,0 @@ -1 一 -2 二 -3 三 -4 四 -5 五 -6 六 -7 七 -8 八 -9 九 -10 十 -11 十一 -12 十二 -13 十三 -14 十四 -15 十五 -16 十六 -17 十七 -18 十八 -19 十九 -20 二十 -21 二十一 -22 二十二 -23 二十三 -24 二十四 -25 二十五 -26 二十六 -27 二十七 -28 二十八 -29 二十九 -30 三十 -31 三十一 -32 三十二 -33 三十三 -34 三十四 -35 三十五 -36 三十六 -37 三十七 -38 三十八 -39 三十九 -40 四十 -41 四十一 -42 四十二 -43 四十三 -44 四十四 -45 四十五 -46 四十六 -47 四十七 -48 四十八 -49 四十九 -50 五十 -51 五十一 -52 五十二 -53 五十三 -54 五十四 -55 五十五 -56 五十六 -57 五十七 -58 五十八 -59 五十九 -60 六十 \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/ja/data/time/second.tsv b/nemo_text_processing/text_normalization/ja/data/time/minute_second.tsv similarity index 98% rename from nemo_text_processing/text_normalization/ja/data/time/second.tsv rename to nemo_text_processing/text_normalization/ja/data/time/minute_second.tsv index 5e8276a00..79ac794c4 100644 --- a/nemo_text_processing/text_normalization/ja/data/time/second.tsv +++ b/nemo_text_processing/text_normalization/ja/data/time/minute_second.tsv @@ -57,4 +57,4 @@ 57 五十七 58 五十八 59 五十九 -60 六十 \ No newline at end of file +60 六十 diff --git a/nemo_text_processing/text_normalization/ja/data/time/suffix.tsv b/nemo_text_processing/text_normalization/ja/data/time/suffix.tsv new file mode 100644 index 000000000..71bc03f55 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/time/suffix.tsv @@ -0,0 +1,8 @@ +hour 時 +duration_hour 時間 +approximate_hour 時頃 +minute 分 +past 過ぎ +approximate 頃 +half 半 +second 秒 diff --git a/nemo_text_processing/text_normalization/ja/data/whitelist.tsv b/nemo_text_processing/text_normalization/ja/data/whitelist.tsv index d0d7bef70..203f6396d 100644 --- a/nemo_text_processing/text_normalization/ja/data/whitelist.tsv +++ b/nemo_text_processing/text_normalization/ja/data/whitelist.tsv @@ -1,17 +1,7 @@ -Dr. ドクター -dr. ドクター -Mr. ミスター -mr. ミスター -Ms. ミス -ms. ミス -Mrs. ミシーズ -mrs. ミシーず -st. ストリード -St. ストリード +st. ストリート +St. ストリート mt. マウント Mt. マウント -Prof. プロフェッサー -prof. プロフェッサー sr. シニア Sr. シニア jr. ジュニア diff --git a/nemo_text_processing/text_normalization/ja/data/whitelist_title.tsv b/nemo_text_processing/text_normalization/ja/data/whitelist_title.tsv new file mode 100644 index 000000000..649cabe0d --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/data/whitelist_title.tsv @@ -0,0 +1,10 @@ +Dr. ドクター +dr. ドクター +Mr. ミスター +mr. ミスター +Mrs. ミセス +mrs. ミセス +Ms. ミス +ms. ミス +Prof. プロフェッサー +prof. プロフェッサー diff --git a/nemo_text_processing/text_normalization/ja/taggers/address.py b/nemo_text_processing/text_normalization/ja/taggers/address.py new file mode 100644 index 000000000..32a18dcd1 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/taggers/address.py @@ -0,0 +1,64 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.text_normalization.ja.graph_utils import NEMO_DIGIT, NEMO_NOT_SPACE, GraphFst +from nemo_text_processing.text_normalization.ja.utils import get_abs_path + + +class AddressFst(GraphFst): + """ + Finite state transducer for classifying Japanese address-like expressions. + + Examples: + 東京都千代田区丸の内1-1-1 -> name: "東京都千代田区丸の内一の一の一" + 503号室 -> name: "五〇三号室" + 〒100-0001 -> name: "郵便番号一ゼロゼロのゼロゼロゼロ一" + """ + + def __init__(self, cardinal: GraphFst, deterministic: bool = True): + super().__init__(name="address", kind="classify", deterministic=deterministic) + + digit = pynini.string_file(get_abs_path("data/numbers/digit.tsv")) + zero = pynini.string_file(get_abs_path("data/numbers/zero.tsv")) + zero_maru = pynini.string_file(get_abs_path("data/numbers/zero_maru.tsv")) + + address_number = cardinal.just_cardinals + digit_for_room = digit | zero_maru + digit_for_postal = digit | zero + separator = pynini.string_file(get_abs_path("data/address/separator.tsv")) + separator_input = pynini.project(separator, "input") + postal = pynini.string_file(get_abs_path("data/address/postal.tsv")) + room_suffix = pynini.string_file(get_abs_path("data/address/room_suffix.tsv")) + + hyphen_to_no = ( + pynini.closure(pynutil.delete(" "), 0, 1) + separator + pynini.closure(pynutil.delete(" "), 0, 1) + ) + + address_chain = address_number + hyphen_to_no + address_number + hyphen_to_no + address_number + + address_prefix_char = pynini.difference( + NEMO_NOT_SPACE, + NEMO_DIGIT | separator_input, + ) + address_with_prefix = pynini.closure(address_prefix_char, 1) + address_chain + + postal_code = postal + digit_for_postal**3 + hyphen_to_no + digit_for_postal**4 + + room = (NEMO_DIGIT**3 @ (digit_for_room**3)) + room_suffix + + graph = address_with_prefix | postal_code | room + self.fst = (pynutil.insert('name: "') + graph + pynutil.insert('"')).optimize() diff --git a/nemo_text_processing/text_normalization/ja/taggers/cardinal.py b/nemo_text_processing/text_normalization/ja/taggers/cardinal.py index ff80f6a3b..86e7a22ff 100644 --- a/nemo_text_processing/text_normalization/ja/taggers/cardinal.py +++ b/nemo_text_processing/text_normalization/ja/taggers/cardinal.py @@ -17,7 +17,7 @@ from pynini.lib import pynutil from nemo_text_processing.text_normalization.ja.graph_utils import NEMO_DIGIT, GraphFst -from nemo_text_processing.text_normalization.ja.utils import get_abs_path +from nemo_text_processing.text_normalization.ja.utils import get_abs_path, load_labels class CardinalFst(GraphFst): @@ -38,35 +38,45 @@ def __init__(self, deterministic: bool = True): graph_digit_alt = no_zero_and_one @ graph_digit graph_ties = pynini.string_file(get_abs_path("data/numbers/ties.tsv")) graph_teen = pynini.string_file(get_abs_path("data/numbers/teen.tsv")) + magnitudes = dict(load_labels(get_abs_path("data/numbers/magnitude.tsv"))) + hundred = magnitudes["hundred"] + thousand = magnitudes["thousand"] + ten_thousand = magnitudes["ten_thousand"] + hundred_million = magnitudes["hundred_million"] - graph_all = (graph_ties + (graph_digit | pynutil.delete('0'))) | graph_teen | graph_digit + graph_all = (graph_ties + (graph_digit | pynutil.delete("0"))) | graph_teen | graph_digit hundreds = NEMO_DIGIT**3 - graph_hundred_component = (pynini.cross('1', '百') | (graph_digit_alt + pynutil.insert('百'))) + pynini.union( - pynini.closure(pynutil.delete('0')), (pynini.closure(pynutil.delete('0')) + graph_all) + graph_hundred_component = ( + pynini.cross("1", hundred) | (graph_digit_alt + pynutil.insert(hundred)) + ) + pynini.union( + pynini.closure(pynutil.delete("0")), + (pynini.closure(pynutil.delete("0")) + graph_all), ) graph_hundred = hundreds @ graph_hundred_component thousands = NEMO_DIGIT**4 - graph_thousand_component = (pynini.cross('1', '千') | (graph_digit_alt + pynutil.insert('千'))) + pynini.union( - pynini.closure(pynutil.delete('0')), + graph_thousand_component = ( + pynini.cross("1", thousand) | (graph_digit_alt + pynutil.insert(thousand)) + ) + pynini.union( + pynini.closure(pynutil.delete("0")), graph_hundred_component, - (pynini.closure(pynutil.delete('0')) + graph_all), + (pynini.closure(pynutil.delete("0")) + graph_all), ) - graph_thousand_component_alt = (graph_digit + pynutil.insert('千')) + pynini.union( - pynini.closure(pynutil.delete('0')), + graph_thousand_component_alt = (graph_digit + pynutil.insert(thousand)) + pynini.union( + pynini.closure(pynutil.delete("0")), graph_hundred_component, - (pynini.closure(pynutil.delete('0')) + graph_all), + (pynini.closure(pynutil.delete("0")) + graph_all), ) # this grammar is for larger number in later gramamr graph_thousand = thousands @ graph_thousand_component ten_thousands = NEMO_DIGIT**5 - graph_ten_thousand_component = (graph_digit + pynutil.insert('万')) + pynini.union( - pynini.closure(pynutil.delete('0')), + graph_ten_thousand_component = (graph_digit + pynutil.insert(ten_thousand)) + pynini.union( + pynini.closure(pynutil.delete("0")), graph_thousand_component, - (pynutil.delete('0') + graph_hundred_component), - (pynini.closure(pynutil.delete('0')) + graph_all), + (pynutil.delete("0") + graph_hundred_component), + (pynini.closure(pynutil.delete("0")) + graph_all), ) graph_ten_thousand = ten_thousands @ graph_ten_thousand_component self.man = graph_ten_thousand.optimize() @@ -74,91 +84,95 @@ def __init__(self, deterministic: bool = True): hundred_thousands = NEMO_DIGIT**6 hundred_thousands_position = NEMO_DIGIT**2 hundred_thousands_position = hundred_thousands_position @ graph_all - graph_hundred_thousand_component = (hundred_thousands_position + pynutil.insert('万')) + pynini.union( - pynini.closure(pynutil.delete('0')), + graph_hundred_thousand_component = (hundred_thousands_position + pynutil.insert(ten_thousand)) + pynini.union( + pynini.closure(pynutil.delete("0")), graph_thousand_component, - (pynutil.delete('0') + graph_hundred_component), - (pynini.closure(pynutil.delete('0')) + graph_all), + (pynutil.delete("0") + graph_hundred_component), + (pynini.closure(pynutil.delete("0")) + graph_all), ) graph_hundred_thousand = hundred_thousands @ graph_hundred_thousand_component millions = NEMO_DIGIT**7 million_position = NEMO_DIGIT**3 million_position = million_position @ graph_hundred_component - graph_million_component = (million_position + pynutil.insert('万')) + pynini.union( - pynini.closure(pynutil.delete('0')), + graph_million_component = (million_position + pynutil.insert(ten_thousand)) + pynini.union( + pynini.closure(pynutil.delete("0")), graph_thousand_component, - (pynutil.delete('0') + graph_hundred_component), - (pynini.closure(pynutil.delete('0')) + graph_all), + (pynutil.delete("0") + graph_hundred_component), + (pynini.closure(pynutil.delete("0")) + graph_all), ) graph_million = millions @ graph_million_component ten_millions = NEMO_DIGIT**8 ten_million_position = NEMO_DIGIT**4 ten_million_position = ten_million_position @ graph_thousand_component_alt - graph_ten_million_component = (ten_million_position + pynutil.insert('万')) + pynini.union( - pynini.closure(pynutil.delete('0')), + graph_ten_million_component = (ten_million_position + pynutil.insert(ten_thousand)) + pynini.union( + pynini.closure(pynutil.delete("0")), graph_thousand_component, - (pynutil.delete('0') + graph_hundred_component), - (pynini.closure(pynutil.delete('0')) + graph_all), + (pynutil.delete("0") + graph_hundred_component), + (pynini.closure(pynutil.delete("0")) + graph_all), ) graph_ten_million = ten_millions @ graph_ten_million_component hundred_millions = NEMO_DIGIT**9 - graph_hundred_million_component = (graph_digit + pynutil.insert('億')) + pynini.union( - pynini.closure(pynutil.delete('0')), + graph_hundred_million_component = (graph_digit + pynutil.insert(hundred_million)) + pynini.union( + pynini.closure(pynutil.delete("0")), graph_ten_million_component, - (pynutil.delete('0') + graph_million_component), - (pynutil.delete('00') + graph_hundred_thousand_component), - (pynutil.delete('000') + graph_ten_thousand_component), - (pynutil.delete('0000') + graph_thousand_component), - ((pynutil.delete('00000') + graph_hundred_component)), - (pynini.closure(pynutil.delete('0')) + graph_all), + (pynutil.delete("0") + graph_million_component), + (pynutil.delete("00") + graph_hundred_thousand_component), + (pynutil.delete("000") + graph_ten_thousand_component), + (pynutil.delete("0000") + graph_thousand_component), + ((pynutil.delete("00000") + graph_hundred_component)), + (pynini.closure(pynutil.delete("0")) + graph_all), ) graph_hundred_million = hundred_millions @ graph_hundred_million_component thousand_millions = NEMO_DIGIT**10 thousand_millions_position = NEMO_DIGIT**2 thousand_millions_position = thousand_millions_position @ graph_all - graph_thousand_million_component = (thousand_millions_position + pynutil.insert('億')) + pynini.union( - pynini.closure(pynutil.delete('0')), + graph_thousand_million_component = ( + thousand_millions_position + pynutil.insert(hundred_million) + ) + pynini.union( + pynini.closure(pynutil.delete("0")), graph_ten_million_component, - (pynutil.delete('0') + graph_million_component), - (pynutil.delete('00') + graph_hundred_thousand_component), - (pynutil.delete('000') + graph_ten_thousand_component), - (pynutil.delete('0000') + graph_thousand_component), - ((pynutil.delete('00000') + graph_hundred_component)), - (pynini.closure(pynutil.delete('0')) + graph_all), + (pynutil.delete("0") + graph_million_component), + (pynutil.delete("00") + graph_hundred_thousand_component), + (pynutil.delete("000") + graph_ten_thousand_component), + (pynutil.delete("0000") + graph_thousand_component), + ((pynutil.delete("00000") + graph_hundred_component)), + (pynini.closure(pynutil.delete("0")) + graph_all), ) graph_thousand_million = thousand_millions @ graph_thousand_million_component ten_billions = NEMO_DIGIT**11 ten_billions_position = NEMO_DIGIT**3 ten_billions_position = ten_billions_position @ graph_hundred_component - graph_ten_billions_component = (ten_billions_position + pynutil.insert('億')) + pynini.union( - pynini.closure(pynutil.delete('0')), + graph_ten_billions_component = (ten_billions_position + pynutil.insert(hundred_million)) + pynini.union( + pynini.closure(pynutil.delete("0")), graph_ten_million_component, - (pynutil.delete('0') + graph_million_component), - (pynutil.delete('00') + graph_hundred_thousand_component), - (pynutil.delete('000') + graph_ten_thousand_component), - (pynutil.delete('0000') + graph_thousand_component), - ((pynutil.delete('00000') + graph_hundred_component)), - (pynini.closure(pynutil.delete('0')) + graph_all), + (pynutil.delete("0") + graph_million_component), + (pynutil.delete("00") + graph_hundred_thousand_component), + (pynutil.delete("000") + graph_ten_thousand_component), + (pynutil.delete("0000") + graph_thousand_component), + ((pynutil.delete("00000") + graph_hundred_component)), + (pynini.closure(pynutil.delete("0")) + graph_all), ) graph_ten_billions = ten_billions @ graph_ten_billions_component hundred_billions = NEMO_DIGIT**12 hundred_billions_position = NEMO_DIGIT**4 hundred_billions_position = hundred_billions_position @ graph_thousand_component_alt - graph_hundred_billions_component = (hundred_billions_position + pynutil.insert('億')) + pynini.union( - pynini.closure(pynutil.delete('0')), + graph_hundred_billions_component = ( + hundred_billions_position + pynutil.insert(hundred_million) + ) + pynini.union( + pynini.closure(pynutil.delete("0")), graph_ten_million_component, - (pynutil.delete('0') + graph_million_component), - (pynutil.delete('00') + graph_hundred_thousand_component), - (pynutil.delete('000') + graph_ten_thousand_component), - (pynutil.delete('0000') + graph_thousand_component), - ((pynutil.delete('00000') + graph_hundred_component)), - (pynini.closure(pynutil.delete('0')) + graph_all), + (pynutil.delete("0") + graph_million_component), + (pynutil.delete("00") + graph_hundred_thousand_component), + (pynutil.delete("000") + graph_ten_thousand_component), + (pynutil.delete("0000") + graph_thousand_component), + ((pynutil.delete("00000") + graph_hundred_component)), + (pynini.closure(pynutil.delete("0")) + graph_all), ) graph_hundred_billions = hundred_billions @ graph_hundred_billions_component @@ -179,12 +193,14 @@ def __init__(self, deterministic: bool = True): self.just_cardinals = graph.optimize() optional_sign = ( - pynutil.insert("negative: \"") + (pynini.accep("-") | pynini.cross("マイナス", "-")) + pynutil.insert("\"") + pynutil.insert('negative: "') + + pynini.string_file(get_abs_path("data/numbers/sign.tsv")) + + pynutil.insert('"') ) final_graph = ( - optional_sign + pynutil.insert(" ") + pynutil.insert("integer: \"") + graph + pynutil.insert("\"") - ) | (pynutil.insert("integer: \"") + graph + pynutil.insert("\"")) + optional_sign + pynutil.insert(" ") + pynutil.insert('integer: "') + graph + pynutil.insert('"') + ) | (pynutil.insert('integer: "') + graph + pynutil.insert('"')) final_graph = self.add_tokens(final_graph) diff --git a/nemo_text_processing/text_normalization/ja/taggers/date.py b/nemo_text_processing/text_normalization/ja/taggers/date.py index a8a469252..26c5f742c 100644 --- a/nemo_text_processing/text_normalization/ja/taggers/date.py +++ b/nemo_text_processing/text_normalization/ja/taggers/date.py @@ -21,7 +21,7 @@ NEMO_NON_BREAKING_SPACE, GraphFst, ) -from nemo_text_processing.text_normalization.ja.utils import get_abs_path +from nemo_text_processing.text_normalization.ja.utils import get_abs_path, load_labels class DateFst(GraphFst): @@ -70,17 +70,29 @@ def __init__(self, cardinal: GraphFst, deterministic: bool = True): week = pynini.string_file(get_abs_path("data/date/week.tsv")) era = pynini.string_file(get_abs_path("data/date/era.tsv")) era_abbrev = pynini.string_file(get_abs_path("data/date/era_abbrev.tsv")) + suffixes = dict(load_labels(get_abs_path("data/date/suffix.tsv"))) + year_suffix = suffixes["year"] + month_suffix = suffixes["month"] + day_suffix = suffixes["day"] + century_suffix = suffixes["century"] + decade_suffix = suffixes["decade"] + ten_day_period = pynini.union( + suffixes["early_ten_days"], + suffixes["middle_ten_days"], + suffixes["late_ten_days"], + ) + range_separator = pynini.string_file(get_abs_path("data/punctuation/range.tsv")) signs = pynutil.delete("/") | pynutil.delete(".") | pynutil.delete("-") delete_spaces = pynini.closure( pynutil.delete(" ") | pynutil.delete(NEMO_NARROW_NON_BREAK_SPACE) | pynutil.delete(NEMO_NON_BREAKING_SPACE) ) - era_component = pynutil.insert("era: \"") + era + pynutil.insert("\"") - era_abbrev_component = pynutil.insert("era: \"") + era_abbrev + pynutil.insert("\"") - year_component = pynutil.insert("year: \"") + graph_cardinal + pynutil.insert("年") + pynutil.insert("\"") - month_component = pynutil.insert("month: \"") + month + pynutil.insert("月") + pynutil.insert("\"") - day_component = pynutil.insert("day: \"") + day + pynutil.insert("日") + pynutil.insert("\"") + era_component = pynutil.insert('era: "') + era + pynutil.insert('"') + era_abbrev_component = pynutil.insert('era: "') + era_abbrev + pynutil.insert('"') + year_component = pynutil.insert('year: "') + graph_cardinal + pynutil.insert(year_suffix) + pynutil.insert('"') + month_component = pynutil.insert('month: "') + month + pynutil.insert(month_suffix) + pynutil.insert('"') + day_component = pynutil.insert('day: "') + day + pynutil.insert(day_suffix) + pynutil.insert('"') front_bracket = ( ( @@ -119,24 +131,24 @@ def __init__(self, cardinal: GraphFst, deterministic: bool = True): # this graph optionally accepts () around weekday to accomodate to inputs like (月〜金), thus being longer week_component = ( - (front_bracket + pynutil.insert("weekday: \"") + week + preceding_bracket + pynutil.insert("\"")) + (front_bracket + pynutil.insert('weekday: "') + week + preceding_bracket + pynutil.insert('"')) | ( front_bracket - + pynutil.insert("weekday: \"") + + pynutil.insert('weekday: "') + week - + pynini.cross("〜", "から") + + range_separator + week + preceding_bracket - + pynutil.insert("\"") + + pynutil.insert('"') ) | ( front_bracket - + pynutil.insert("weekday: \"") + + pynutil.insert('weekday: "') + week + pynutil.delete("・") + week + preceding_bracket - + pynutil.insert("\"") + + pynutil.insert('"') ) ) @@ -154,37 +166,33 @@ def __init__(self, cardinal: GraphFst, deterministic: bool = True): # 2024年, 9月, 28日 individual_year_component = ( pynini.closure(era_component + pynutil.insert(" "), 0, 1) - + pynutil.insert("year: \"") + + pynutil.insert('year: "') + graph_cardinal - + pynini.accep("年") - + pynutil.insert("\"") + + pynini.accep(year_suffix) + + pynutil.insert('"') ) # this extra individual year component is to accomodate inputs R. 2024 with out "年" # the inputs may or maynot include "年", thus below: individual_year_component_2 = ( pynini.closure(era_component + pynutil.insert(" "), 0, 1) - + pynutil.insert("year: \"") + + pynutil.insert('year: "') + graph_cardinal - + (pynini.accep("世紀") | pynini.accep("")) - + pynutil.insert("\"") + + (pynini.accep(century_suffix) | pynini.accep("")) + + pynutil.insert('"') ) | ( era_abbrev_component + pynutil.insert(" ") - + pynutil.insert("year: \"") + + pynutil.insert('year: "') + graph_cardinal - + pynutil.insert("年") - + pynutil.insert("\"") + + pynutil.insert(year_suffix) + + pynutil.insert('"') ) individual_month_component = ( - pynutil.insert("month: \"") + month + pynini.accep("月") + pynutil.insert("\"") - ) | ( - pynutil.insert("month: \"") - + (pynini.accep("中旬") | pynini.accep("下旬") | pynini.accep("上旬")) - + pynutil.insert("\"") - ) + pynutil.insert('month: "') + month + pynini.accep(month_suffix) + pynutil.insert('"') + ) | (pynutil.insert('month: "') + ten_day_period + pynutil.insert('"')) individual_day_component = ( - pynutil.insert("day: \"") + graph_cardinal + pynini.accep("日") + pynutil.insert("\"") + pynutil.insert('day: "') + graph_cardinal + pynini.accep(day_suffix) + pynutil.insert('"') ) graph_individual_component = ( @@ -208,13 +216,13 @@ def __init__(self, cardinal: GraphFst, deterministic: bool = True): ) ) + pynini.closure(pynutil.insert(" ") + week_component, 0, 1) - nendai = pynini.accep("年代") + nendai = pynini.accep(decade_suffix) era_nendai = ( pynini.closure(era_component + pynutil.insert(" "), 0, 1) - + pynutil.insert("year: \"") + + pynutil.insert('year: "') + graph_cardinal + nendai - + pynutil.insert("\"") + + pynutil.insert('"') ) graph_all_date = ( diff --git a/nemo_text_processing/text_normalization/ja/taggers/decimal.py b/nemo_text_processing/text_normalization/ja/taggers/decimal.py index 8fdea4c87..228e59e1c 100644 --- a/nemo_text_processing/text_normalization/ja/taggers/decimal.py +++ b/nemo_text_processing/text_normalization/ja/taggers/decimal.py @@ -16,7 +16,7 @@ import pynini from pynini.lib import pynutil -from nemo_text_processing.text_normalization.ja.graph_utils import GraphFst +from nemo_text_processing.text_normalization.ja.graph_utils import NEMO_DIGIT, GraphFst from nemo_text_processing.text_normalization.ja.utils import get_abs_path @@ -24,6 +24,7 @@ class DecimalFst(GraphFst): """ Finite state transducer for classifying decimal, e.g. 0.5 -> decimal { integer_part: "零" fractional_part: "五" } + 0.05 -> decimal { integer_part: "零" fractional_part: "零五" } -0.5万 -> decimal { negative: "マイナス" integer_part: "零" fractional_part: "五" quantity: "万"} Args: @@ -35,22 +36,31 @@ def __init__(self, cardinal: GraphFst, deterministic: bool = True): cardinal_before_decimal = cardinal.just_cardinals cardinal_after_decimal = pynini.string_file(get_abs_path("data/numbers/digit.tsv")) - zero = pynini.string_file(get_abs_path("data/numbers/zero.tsv")) + zero_decimal = pynini.string_file(get_abs_path("data/numbers/zero_decimal.tsv")) + decimal_point = pynini.string_file(get_abs_path("data/numbers/decimal_point.tsv")) + decimal_separator = pynini.project(decimal_point, "input") + sign = pynini.string_file(get_abs_path("data/numbers/sign.tsv")) - graph_integer = pynutil.insert('integer_part: \"') + cardinal_before_decimal + pynutil.insert("\"") + graph_integer = ( + pynutil.insert('integer_part: "') + + ( + zero_decimal + | (pynini.difference(NEMO_DIGIT, "0") @ cardinal_before_decimal) + | (pynini.closure(NEMO_DIGIT, 2) @ cardinal_before_decimal) + ) + + pynutil.insert('"') + ) graph_fraction = ( - pynutil.insert("fractional_part: \"") - + pynini.closure((cardinal_after_decimal | zero), 1) - + pynutil.insert("\"") + pynutil.insert('fractional_part: "') + + pynini.closure((cardinal_after_decimal | zero_decimal), 1) + + pynutil.insert('"') ) - graph_decimal_no_sign = graph_integer + pynutil.delete('.') + pynutil.insert(" ") + graph_fraction - - graph_optional_sign = ( - pynutil.insert("negative: \"") - + (pynini.cross("-", "マイナス") | pynini.accep("マイナス")) - + pynutil.insert("\"") + graph_decimal_no_sign = ( + graph_integer + pynutil.delete(decimal_separator) + pynutil.insert(" ") + graph_fraction ) + graph_optional_sign = pynutil.insert('negative: "') + sign + pynutil.insert('"') + graph_decimal = graph_decimal_no_sign | (graph_optional_sign + pynutil.insert(" ") + graph_decimal_no_sign) self.just_decimal = graph_decimal_no_sign.optimize() diff --git a/nemo_text_processing/text_normalization/ja/taggers/electronic.py b/nemo_text_processing/text_normalization/ja/taggers/electronic.py new file mode 100644 index 000000000..2c61ad6e3 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/taggers/electronic.py @@ -0,0 +1,129 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.text_normalization.ja.graph_utils import NEMO_ALPHA, NEMO_DIGIT, NEMO_NOT_SPACE, GraphFst +from nemo_text_processing.text_normalization.ja.utils import get_abs_path + + +class ElectronicFst(GraphFst): + """ + Finite state transducer for classifying Japanese electronic expressions. + + Examples: + abc@abc.com -> electronic { username: "abc" domain: "abc.com" preserve_order: true } + https://www.nvidia.com + -> electronic { protocol: "https" domain: "www.nvidia.com" preserve_order: true } + 1234-5678-9012-3456 + -> electronic { domain: "1234 5678 9012 3456" preserve_order: true } + """ + + def __init__(self, cardinal: GraphFst, deterministic: bool = True): + super().__init__(name="electronic", kind="classify", deterministic=deterministic) + + alnum = NEMO_ALPHA | NEMO_DIGIT + hyphen = pynini.accep("-") + dot = pynini.accep(".") + slash = pynini.accep("/") + at = pynini.accep("@") + + label = pynini.closure(alnum | hyphen, 1) + tld = pynini.closure(NEMO_ALPHA, 2) + domain_core = label + pynini.closure(dot + label) + dot + tld + domain_field = pynutil.insert('domain: "') + domain_core + pynutil.insert('"') + + username_symbol = dot | hyphen + username_core = alnum + pynini.closure(alnum | username_symbol) + username_field = ( + pynutil.insert('username: "') + + username_core + + pynutil.insert('"') + + pynutil.delete("@") + + pynutil.insert(" ") + ) + email = username_field + domain_field + + protocol = pynini.string_file(get_abs_path("data/electronic/protocol.tsv")) + protocol_field = pynutil.insert('protocol: "') + protocol + pynutil.insert('"') + path_segment = pynini.closure(alnum | hyphen, 1) + path_core = slash + path_segment + pynini.closure(slash + path_segment) + path_field = pynutil.insert(' path: "') + path_core + pynutil.insert('"') + url = ( + protocol_field + + pynutil.delete("://") + + pynutil.insert(" ") + + domain_field + + pynini.closure(path_field, 0, 1) + ) + + four_digits = NEMO_DIGIT**4 + card_separator = pynutil.delete("-") | pynutil.delete(" ") + grouped_card_number = ( + four_digits + + card_separator + + pynutil.insert(" ") + + four_digits + + card_separator + + pynutil.insert(" ") + + four_digits + + card_separator + + pynutil.insert(" ") + + four_digits + ) + card_number_field = pynutil.insert('domain: "') + grouped_card_number + pynutil.insert('"') + short_card_number_field = pynutil.insert('domain: "') + four_digits + pynutil.insert('"') + + card_cue = pynini.string_file(get_abs_path("data/electronic/card_cues.tsv")) + card_cue_field = pynutil.insert('protocol: "') + card_cue + pynutil.insert('" ') + credit_card = card_number_field + card_with_cue = card_cue_field + card_number_field + card_tail_with_cue = card_cue_field + short_card_number_field + + digit_count_prefix = pynini.string_file(get_abs_path("data/electronic/card_digit_count_prefix.tsv")) + digit_count_suffix = pynini.string_file(get_abs_path("data/electronic/card_digit_count_suffix.tsv")) + digit = pynini.string_file(get_abs_path("data/numbers/digit.tsv")) + card_tail = ( + pynutil.insert('protocol: "') + + digit_count_prefix + + digit + + digit_count_suffix + + pynutil.insert('" ') + + short_card_number_field + ) + + extension = pynini.project( + pynini.string_file(get_abs_path("data/electronic/file_extensions.tsv")), + "input", + ) + filename_stem = pynini.closure( + pynini.difference(NEMO_NOT_SPACE, pynini.union(dot, slash, at)), + 1, + ) + filename = pynutil.insert('domain: "') + filename_stem + extension + pynutil.insert('"') + + graph = ( + pynutil.add_weight(credit_card, -0.1) + | pynutil.add_weight(card_with_cue, -0.1) + | pynutil.add_weight(card_tail_with_cue, -0.1) + | card_tail + | email + | url + | domain_field + | filename + ) + graph += pynutil.insert(" preserve_order: true") + + self.fst = self.add_tokens(graph.optimize()).optimize() diff --git a/nemo_text_processing/text_normalization/ja/taggers/fraction.py b/nemo_text_processing/text_normalization/ja/taggers/fraction.py index 94fb4af68..a0433f97e 100644 --- a/nemo_text_processing/text_normalization/ja/taggers/fraction.py +++ b/nemo_text_processing/text_normalization/ja/taggers/fraction.py @@ -17,7 +17,7 @@ from pynini.lib import pynutil from nemo_text_processing.text_normalization.ja.graph_utils import NEMO_SPACE, GraphFst -from nemo_text_processing.text_normalization.ja.utils import get_abs_path +from nemo_text_processing.text_normalization.ja.utils import get_abs_path, load_labels class FractionFst(GraphFst): @@ -25,7 +25,7 @@ class FractionFst(GraphFst): Finite state transducer for classifying fractions, e.g. 1/2 -> tokens { fraction { denominator: "二" numerator: "一"} } 1と3/4 -> fraction { integer: "一" denominator: "四" numerator: "三" } - 一荷四分の三 -> fraction { integer: "1" denominator: "4" numerator: "3" } + 一と四分の三 -> fraction { integer: "1" denominator: "4" numerator: "3" } ルート三分の一 -> fraction { denominator: "√3" numerator: "1" } 一点六五分の五十 -> fraction { denominator: "1.65" numerator: "50" } マイナス1/2 -> tokens { fraction { denominator: "二" numerator: "一"} } @@ -40,40 +40,39 @@ def __init__(self, cardinal: GraphFst, deterministic: bool = True): cardinal = cardinal.just_cardinals graph_digit = pynini.string_file(get_abs_path("data/numbers/digit.tsv")) graph_zero = pynini.string_file(get_abs_path("data/numbers/zero.tsv")) + markers = dict(load_labels(get_abs_path("data/fraction/marker.tsv"))) + decimal_point = pynini.string_file(get_abs_path("data/numbers/decimal_point.tsv")) + sign = pynini.string_file(get_abs_path("data/numbers/sign.tsv")) - slash = pynutil.delete('/') - morphemes = pynini.accep('分の') - root = pynini.accep('√') + slash = pynutil.delete("/") + morphemes = pynini.accep(markers["fraction"]) + root = pynini.accep(markers["root_written"]) + mixed = pynini.accep(markers["mixed"]) decimal_number = ( - cardinal - + pynini.cross(".", "点") - + pynini.closure(pynini.closure(graph_digit) | pynini.closure(graph_zero)) + cardinal + decimal_point + pynini.closure(pynini.closure(graph_digit) | pynini.closure(graph_zero)) ) integer_component = ( - pynutil.insert('integer_part: \"') + pynutil.insert('integer_part: "') + (cardinal | (root + cardinal) | decimal_number | (root + decimal_number)) - + pynutil.insert("\"") + + pynutil.insert('"') ) integer_component_with_char = ( - pynutil.insert('integer_part: \"') - + ( - (cardinal | (root + cardinal) | decimal_number | (root + decimal_number)) - + (pynini.accep("と") | pynini.accep("荷")) - ) - + pynutil.insert("\"") + pynutil.insert('integer_part: "') + + ((cardinal | (root + cardinal) | decimal_number | (root + decimal_number)) + mixed) + + pynutil.insert('"') + pynutil.insert(NEMO_SPACE) ) denominator_component = ( - pynutil.insert("denominator: \"") + pynutil.insert('denominator: "') + (cardinal | (root + cardinal) | decimal_number | (root + decimal_number)) - + pynutil.insert("\"") + + pynutil.insert('"') ) numerator_component = ( - pynutil.insert("numerator: \"") + pynutil.insert('numerator: "') + (cardinal | (root + cardinal) | decimal_number | (root + decimal_number)) - + pynutil.insert("\"") + + pynutil.insert('"') ) # 3/4, 1 3/4, 1と3/4, -3/4, -1 3/4, 1と3/4, √1と3/4 and any combination of root number, cardinal number and decimal number @@ -102,22 +101,17 @@ def __init__(self, cardinal: GraphFst, deterministic: bool = True): ) + denominator_component + pynutil.insert(NEMO_SPACE) - + pynutil.insert("morphosyntactic_features: \"") + + pynutil.insert('morphosyntactic_features: "') + morphemes - + pynutil.insert("\"") + + pynutil.insert('"') + pynutil.insert(NEMO_SPACE) + numerator_component ) - optional_sign = ( - pynutil.insert("negative: \"") - + (pynini.accep("マイナス") | pynini.cross("-", "マイナス")) - + pynutil.insert("\"") - ) + optional_sign = pynutil.insert('negative: "') + sign + pynutil.insert('"') - graph_fraction_slash_sigh = pynini.closure(optional_sign + pynutil.insert(NEMO_SPACE), 0, 1) + ( - graph_fraction_slash | graph_fraction_word - ) + self.graph = (graph_fraction_slash | graph_fraction_word).optimize() + graph_fraction_slash_sigh = pynini.closure(optional_sign + pynutil.insert(NEMO_SPACE), 0, 1) + self.graph graph = graph_fraction_slash_sigh # | diff --git a/nemo_text_processing/text_normalization/ja/taggers/measure.py b/nemo_text_processing/text_normalization/ja/taggers/measure.py new file mode 100644 index 000000000..5fd778d5a --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/taggers/measure.py @@ -0,0 +1,133 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.text_normalization.ja.graph_utils import GraphFst, delete_space +from nemo_text_processing.text_normalization.ja.utils import get_abs_path, load_labels + + +class MeasureFst(GraphFst): + """ + Finite state transducer for classifying Japanese measure expressions. + + Examples: + 5kg -> measure { cardinal { integer: "五" } units: "キロ" preserve_order: true } + 0kg -> measure { cardinal { integer: "ゼロ" } units: "キロ" preserve_order: true } + 0.05m + -> measure { decimal { integer_part: "零" fractional_part: "零五" } units: "メートル" preserve_order: true } + 60km/h -> measure { cardinal { integer: "時速六十" } units: "キロ" preserve_order: true } + 50m/s -> measure { cardinal { integer: "秒速五十" } units: "メートル" preserve_order: true } + + Args: + cardinal: CardinalFst + decimal: DecimalFst + fraction: FractionFst + deterministic: if True provides a single transduction option + """ + + def __init__( + self, + cardinal: GraphFst, + decimal: GraphFst, + fraction: GraphFst, + deterministic: bool = True, + ): + super().__init__(name="measure", kind="classify", deterministic=deterministic) + + unit_path = get_abs_path("data/measure/unit.tsv") + per_unit_path = get_abs_path("data/measure/per_unit.tsv") + rate_numerator_path = get_abs_path("data/measure/rate_numerator.tsv") + unit = pynini.string_file(unit_path) + rate_numerator_labels = load_labels(rate_numerator_path) + per_unit_labels = load_labels(per_unit_path) + speed_configs = { + (unit_spoken, per_spoken): prefix + for unit_spoken, per_spoken, prefix in load_labels(get_abs_path("data/measure/speed.tsv")) + } + per_marker = load_labels(get_abs_path("data/measure/per_marker.tsv"))[0][0] + + slash = pynutil.delete("/") | pynutil.delete("/") + + general_per_unit = pynini.Fst() + for unit_written, unit_spoken in load_labels(unit_path): + for per_written, per_spoken in per_unit_labels: + general_per_unit |= ( + pynini.cross(unit_written, unit_spoken) + + delete_space + + slash + + delete_space + + pynutil.insert(per_marker) + + pynini.cross(per_written, per_spoken) + ) + unit_graph = unit | general_per_unit + + unit_component = delete_space + pynutil.insert(' units: "') + unit_graph + pynutil.insert('"') + + optional_sign = ( + pynutil.insert('negative: "') + + pynini.string_file(get_abs_path("data/numbers/sign.tsv")) + + pynutil.insert('" ') + + delete_space + ) + + cardinal_graph = ( + pynutil.insert("cardinal { ") + + pynini.closure(optional_sign, 0, 1) + + pynutil.insert('integer: "') + + cardinal.just_cardinals + + pynutil.insert('" }') + ) + decimal_graph = ( + pynutil.insert("decimal { ") + + pynini.closure(optional_sign, 0, 1) + + decimal.just_decimal + + pynutil.insert(" }") + ) + fraction_graph = ( + pynutil.insert("fraction { ") + pynini.closure(optional_sign, 0, 1) + fraction.graph + pynutil.insert(" }") + ) + + number = cardinal_graph | decimal_graph | fraction_graph + + speed_graph = pynini.Fst() + for unit_written, unit_spoken in rate_numerator_labels: + for per_written, per_spoken in per_unit_labels: + prefix = speed_configs.get((unit_spoken, per_spoken)) + if prefix is None: + continue + speed_number = ( + pynutil.insert("cardinal { ") + + pynini.closure(optional_sign, 0, 1) + + pynutil.insert(f'integer: "{prefix}') + + cardinal.just_cardinals + + pynutil.insert('" }') + ) + speed_unit = ( + delete_space + + pynutil.delete(unit_written) + + delete_space + + slash + + delete_space + + pynutil.delete(per_written) + + pynutil.insert(f' units: "{unit_spoken}"') + ) + speed_graph |= speed_number + speed_unit + pynutil.insert(" preserve_order: true") + + general_graph = number + unit_component + pynutil.insert(" preserve_order: true") + + graph = pynutil.add_weight(speed_graph, -0.1) | general_graph + + self.fst = self.add_tokens(graph.optimize()).optimize() diff --git a/nemo_text_processing/text_normalization/ja/taggers/money.py b/nemo_text_processing/text_normalization/ja/taggers/money.py new file mode 100644 index 000000000..52f93cb41 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/taggers/money.py @@ -0,0 +1,151 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.text_normalization.ja.graph_utils import NEMO_DIGIT, GraphFst, delete_space +from nemo_text_processing.text_normalization.ja.utils import get_abs_path, load_labels + + +class MoneyFst(GraphFst): + """ + Finite state transducer for classifying Japanese money expressions. + + Examples: + 100円 -> money { integer_part: "百" currency_maj: "円" preserve_order: true } + ¥3万 -> money { integer_part: "三" quantity: "万" currency_maj: "円" preserve_order: true } + 1.5万円 -> money { integer_part: "一点五" quantity: "万" currency_maj: "円" preserve_order: true } + 5ドル25セント + -> money { integer_part: "五" currency_maj: "ドル" fractional_part: "二十五" + currency_min: "セント" preserve_order: true } + $12.50 + -> money { integer_part: "十二" currency_maj: "ドル" fractional_part: "五十" + currency_min: "セント" preserve_order: true } + + Args: + cardinal: CardinalFst + deterministic: if True will provide a single transduction option + """ + + def __init__(self, cardinal: GraphFst, deterministic: bool = True): + super().__init__(name="money", kind="classify", deterministic=deterministic) + + graph_cardinal = cardinal.just_cardinals + graph_digit = pynini.string_file(get_abs_path("data/numbers/digit.tsv")) + graph_zero = pynini.string_file(get_abs_path("data/numbers/zero.tsv")) + + integer_input = (NEMO_DIGIT + pynini.closure(NEMO_DIGIT | pynutil.delete(","))) @ graph_cardinal + fractional_digits = pynini.closure(graph_digit | graph_zero, 1) + decimal_input = ( + integer_input + pynini.string_file(get_abs_path("data/numbers/decimal_point.tsv")) + fractional_digits + ) + sign_input = pynini.string_file(get_abs_path("data/numbers/sign.tsv")) + delete_space + + integer_component = pynutil.insert('integer_part: "') + integer_input + pynutil.insert('"') + signed_integer_component = ( + pynutil.insert('integer_part: "') + pynini.closure(sign_input, 0, 1) + integer_input + pynutil.insert('"') + ) + signed_decimal_component = ( + pynutil.insert('integer_part: "') + pynini.closure(sign_input, 0, 1) + decimal_input + pynutil.insert('"') + ) + + number_component = signed_decimal_component | signed_integer_component + + quantity = pynini.string_file(get_abs_path("data/money/quantity.tsv")) + quantity_component = delete_space + pynutil.insert(' quantity: "') + quantity + pynutil.insert('"') + + currency_major_labels = load_labels(get_abs_path("data/money/currency_major.tsv")) + currency_major = pynini.string_file(get_abs_path("data/money/currency_major.tsv")) + currency_major_component = ( + delete_space + pynutil.insert(' currency_maj: "') + currency_major + pynutil.insert('"') + ) + + currency_prefix_labels = load_labels(get_abs_path("data/money/currency_prefix.tsv")) + currency_minor_by_major = dict(load_labels(get_abs_path("data/money/currency_minor_by_major.tsv"))) + currency_minor = pynini.string_file(get_abs_path("data/money/currency_minor.tsv")) + non_zero_digit = pynini.difference(NEMO_DIGIT, "0") + minor_decimal_input = (NEMO_DIGIT**2 @ graph_cardinal) | (pynutil.delete("0") + (non_zero_digit @ graph_digit)) + + suffix_graph = number_component + pynini.closure(quantity_component, 0, 1) + currency_major_component + for written, spoken in currency_major_labels: + minor_spoken = currency_minor_by_major.get(written) + if not minor_spoken: + continue + + currency_major_suffix = ( + delete_space + pynutil.delete(written) + pynutil.insert(f' currency_maj: "{spoken}"') + ) + minor_suffix = ( + delete_space + + pynutil.insert(' fractional_part: "') + + integer_input + + pynutil.insert('"') + + delete_space + + (currency_minor @ pynini.cross(minor_spoken, "")) + + pynutil.insert(f' currency_min: "{minor_spoken}"') + ) + suffix_graph |= signed_integer_component + currency_major_suffix + minor_suffix + + prefix_graph = pynini.Fst() + for written, spoken in currency_prefix_labels: + currency_prefix = pynutil.delete(written) + delete_space + currency_field = pynutil.insert(f' currency_maj: "{spoken}"') + minor_spoken = currency_minor_by_major.get(written) + + prefix_graph |= ( + currency_prefix + number_component + pynini.closure(quantity_component, 0, 1) + currency_field + ) + prefix_graph |= ( + pynutil.insert('integer_part: "') + + sign_input + + currency_prefix + + (decimal_input | integer_input) + + pynutil.insert('"') + + pynini.closure(quantity_component, 0, 1) + + currency_field + ) + if minor_spoken: + decimal_minor_component = ( + integer_component + + pynutil.delete(".") + + currency_field + + pynutil.insert(' fractional_part: "') + + minor_decimal_input + + pynutil.insert(f'" currency_min: "{minor_spoken}"') + ) + signed_decimal_minor_component = ( + pynutil.insert('integer_part: "') + + sign_input + + currency_prefix + + integer_input + + pynutil.delete(".") + + pynutil.insert('"') + + currency_field + + pynutil.insert(' fractional_part: "') + + minor_decimal_input + + pynutil.insert(f'" currency_min: "{minor_spoken}"') + ) + prefix_graph |= pynutil.add_weight( + currency_prefix + decimal_minor_component, + -0.1, + ) + prefix_graph |= pynutil.add_weight( + signed_decimal_minor_component, + -0.1, + ) + + graph = (suffix_graph | prefix_graph) + pynutil.insert(" preserve_order: true") + + self.fst = self.add_tokens(graph.optimize()).optimize() diff --git a/nemo_text_processing/text_normalization/ja/taggers/ordinal.py b/nemo_text_processing/text_normalization/ja/taggers/ordinal.py index d88608e01..e269615fb 100644 --- a/nemo_text_processing/text_normalization/ja/taggers/ordinal.py +++ b/nemo_text_processing/text_normalization/ja/taggers/ordinal.py @@ -17,6 +17,7 @@ from pynini.lib import pynutil from nemo_text_processing.text_normalization.ja.graph_utils import GraphFst +from nemo_text_processing.text_normalization.ja.utils import get_abs_path, load_labels class OrdinalFst(GraphFst): @@ -32,11 +33,12 @@ def __init__(self, cardinal: GraphFst, deterministic: bool = True): super().__init__(name="ordinal", kind="classify", deterministic=deterministic) graph_cardinal = cardinal.just_cardinals - morpheme_pre = pynini.accep('第') - morpheme_post = pynini.accep('番目') + markers = dict(load_labels(get_abs_path("data/ordinal/marker.tsv"))) + morpheme_pre = pynini.accep(markers["prefix"]) + morpheme_post = pynini.accep(markers["suffix"]) graph_ordinal = pynini.union(morpheme_pre + graph_cardinal, graph_cardinal + morpheme_post) - final_graph = pynutil.insert("integer: \"") + graph_ordinal + pynutil.insert("\"") + final_graph = pynutil.insert('integer: "') + graph_ordinal + pynutil.insert('"') graph_ordinal_final = self.add_tokens(final_graph) self.fst = graph_ordinal_final.optimize() diff --git a/nemo_text_processing/text_normalization/ja/taggers/punctuation.py b/nemo_text_processing/text_normalization/ja/taggers/punctuation.py index c5df8388c..0fb45ea6b 100644 --- a/nemo_text_processing/text_normalization/ja/taggers/punctuation.py +++ b/nemo_text_processing/text_normalization/ja/taggers/punctuation.py @@ -35,20 +35,20 @@ class PunctuationFst(GraphFst): """ - def __init__(self, deterministic: bool = True): - super().__init__(name="punctuation", kind="classify", deterministic=deterministic) - s = "!#$%&'()*+,-./:;<=>?@^_`{|}。,;:《》“”·~【】!?、‘’.<>-——_、。.「」『』‘`/・;’”“”‷・〔〕々〃ゝゞヽ〲〱〳〴〵ヾ〆,~" - - punct_symbols_to_exclude = ["[", "]"] - punct_unicode = [ + def __init__(self, deterministic: bool = True): + super().__init__(name="punctuation", kind="classify", deterministic=deterministic) + + punct_symbols_to_exclude = ["[", "]"] + punct_unicode = [ chr(i) for i in range(sys.maxunicode) - if category(chr(i)).startswith("P") and chr(i) not in punct_symbols_to_exclude - ] - - whitelist_symbols = load_labels(get_abs_path("data/symbol.tsv")) - whitelist_symbols = [x[0] for x in whitelist_symbols] - self.punct_marks = [p for p in punct_unicode + list(s) if p not in whitelist_symbols] + if category(chr(i)).startswith("P") and chr(i) not in punct_symbols_to_exclude + ] + extra_symbols = [label[0] for label in load_labels(get_abs_path("data/punctuation/extra.tsv"))] + + whitelist_symbols = load_labels(get_abs_path("data/symbol.tsv")) + whitelist_symbols = [x[0] for x in whitelist_symbols] + self.punct_marks = [p for p in punct_unicode + extra_symbols if p not in whitelist_symbols] punct = pynini.union(*self.punct_marks) punct = pynini.closure(punct, 1) @@ -62,9 +62,7 @@ def __init__(self, deterministic: bool = True): + pynini.accep(">") ) punct = plurals._priority_union(emphasis, punct, NEMO_SIGMA) - range_component = pynini.cross("〜", "から") | pynini.accep( - "から" - ) # forcing this conversion for special tilde + range_component = pynini.string_file(get_abs_path("data/punctuation/range.tsv")) - self.graph = punct | pynutil.add_weight(range_component, -1.0) - self.fst = (pynutil.insert("name: \"") + self.graph + pynutil.insert("\"")).optimize() + self.graph = plurals._priority_union(range_component, punct, NEMO_SIGMA) + self.fst = (pynutil.insert('name: "') + self.graph + pynutil.insert('"')).optimize() diff --git a/nemo_text_processing/text_normalization/ja/taggers/range.py b/nemo_text_processing/text_normalization/ja/taggers/range.py new file mode 100644 index 000000000..94936fb22 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/taggers/range.py @@ -0,0 +1,71 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.text_normalization.ja.graph_utils import NEMO_DIGIT, GraphFst, delete_space +from nemo_text_processing.text_normalization.ja.utils import get_abs_path + + +class RangeFst(GraphFst): + """ + Finite state transducer for classifying Japanese ranges. + + Examples: + 2-5 -> tokens { name: "二から五" } + 10:00-11:00 -> tokens { name: "十時から十一時" } + 3kg-6kg -> tokens { name: "三キロから六キロ" } + + Args: + cardinal: composed cardinal tagger and verbalizer + date: composed date tagger and verbalizer + time: composed time tagger and verbalizer + money: composed money tagger and verbalizer + measure: composed measure tagger and verbalizer + deterministic: if True will provide a single transduction option, + for False multiple transductions are generated + """ + + def __init__( + self, + cardinal: GraphFst, + date: GraphFst, + time: GraphFst, + money: GraphFst, + measure: GraphFst, + deterministic: bool = True, + ): + super().__init__(name="range", kind="classify", deterministic=deterministic) + + separator = pynini.string_file(get_abs_path("data/range/separator.tsv")) + sep_to_kara = delete_space + separator + delete_space + + endpoint = cardinal | date | time | money | measure + graph = endpoint + sep_to_kara + endpoint + + # Some range suffixes, such as 人 and 歳, do not have a dedicated + # semiotic class. Keep them as range-specific cardinal patterns. + suffix = pynini.string_file(get_abs_path("data/range/suffix.tsv")) + cardinal_pair = cardinal + sep_to_kara + cardinal + cardinal_range = cardinal_pair + pynini.closure(suffix, 0, 1) + graph |= cardinal_range + graph |= cardinal + pynini.string_file(get_abs_path("data/range/operator.tsv")) + cardinal + + # Normalize English-style decade suffixes before reusing the date graph. + year_alias = (NEMO_DIGIT**4 + pynini.string_file(get_abs_path("data/range/year_suffix.tsv"))) @ date + graph |= pynutil.add_weight(year_alias + sep_to_kara + year_alias, -0.5) + + self.graph = graph.optimize() + self.fst = (pynutil.insert('name: "') + self.graph + pynutil.insert('"')).optimize() diff --git a/nemo_text_processing/text_normalization/ja/taggers/roman.py b/nemo_text_processing/text_normalization/ja/taggers/roman.py new file mode 100644 index 000000000..1c6e018af --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/taggers/roman.py @@ -0,0 +1,68 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.text_normalization.ja.graph_utils import GraphFst, insert_space +from nemo_text_processing.text_normalization.ja.utils import get_abs_path, load_labels + + +class RomanFst(GraphFst): + """ + Finite state transducer for classifying Roman numerals in supported contexts. + + Examples: + 第III章 -> tokens { name: "第三章" } + Chapter IV -> tokens { name: "Chapter 四" } + Century XXI -> tokens { name: "Century 二十一" } + """ + + def __init__(self, cardinal: GraphFst, deterministic: bool = True): + super().__init__(name="roman", kind="classify", deterministic=deterministic) + + roman_values = { + roman: int(value) for roman, value in load_labels(get_abs_path("data/roman/roman_numerals.tsv")) + } + valid_roman_pairs = [] + for number in range(1, 4000): + roman = self._int_to_roman(number, roman_values) + valid_roman_pairs.append((roman, str(number))) + valid_roman_pairs.append((roman.lower(), str(number))) + + roman_to_number = pynini.string_map(valid_roman_pairs).optimize() + roman_to_cardinal = roman_to_number @ cardinal.just_cardinals + + japanese_prefix = pynini.string_file(get_abs_path("data/roman/japanese_prefix.tsv")) + japanese_suffix = pynini.string_file(get_abs_path("data/roman/japanese_suffix.tsv")) + japanese_context = japanese_prefix + roman_to_cardinal + japanese_suffix + + key_cardinal = pynini.union( + *[pynini.accep(x[0]) for x in load_labels(get_abs_path("data/roman/key_cardinal.tsv"))] + ) + cardinal_context = key_cardinal + pynutil.delete(" ") + insert_space + roman_to_cardinal + + graph = japanese_context | cardinal_context + self.fst = (pynutil.insert('name: "') + graph.optimize() + pynutil.insert('"')).optimize() + + @staticmethod + def _int_to_roman(number: int, roman_values: dict) -> str: + value_to_roman = sorted(((value, roman) for roman, value in roman_values.items()), reverse=True) + result = [] + remaining = number + for value, roman in value_to_roman: + while remaining >= value: + result.append(roman) + remaining -= value + return "".join(result) diff --git a/nemo_text_processing/text_normalization/ja/taggers/serial.py b/nemo_text_processing/text_normalization/ja/taggers/serial.py new file mode 100644 index 000000000..7942919e5 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/taggers/serial.py @@ -0,0 +1,89 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.text_normalization.ja.graph_utils import ( + NEMO_DIGIT, + NEMO_SIGMA, + NEMO_UPPER, + TO_UPPER, + GraphFst, + insert_space, +) +from nemo_text_processing.text_normalization.ja.utils import get_abs_path + + +class SerialFst(GraphFst): + """ + Finite state transducer for classifying compact serial/model identifiers. + + Examples: + B2A23C -> name: "ビー 二 エー 二三 シー" + MIG-25/235212-asdg + -> name: "エムアイジー ハイフン 二五 スラッシュ 二三五二一二 ハイフン エーエスディージー" + """ + + def __init__(self, cardinal: GraphFst, deterministic: bool = True): + super().__init__(name="serial", kind="classify", deterministic=deterministic) + + uppercase_letters = pynini.string_file(get_abs_path("data/latin/letters.tsv")) + letters = (NEMO_UPPER | TO_UPPER) @ uppercase_letters + letter_input = pynini.project(letters, "input") + digit = pynini.string_file(get_abs_path("data/numbers/digit.tsv")) | pynini.string_file( + get_abs_path("data/numbers/zero_maru.tsv") + ) + insert_letter_digit_space = pynini.cdrewrite(pynutil.insert(" "), letter_input, NEMO_DIGIT, NEMO_SIGMA) + insert_digit_letter_space = pynini.cdrewrite(pynutil.insert(" "), NEMO_DIGIT, letter_input, NEMO_SIGMA) + alnum_spacing = insert_letter_digit_space @ insert_digit_letter_space + raw_alnum = pynini.closure(letter_input | NEMO_DIGIT, 1) + raw_letter_leading_alnum = ( + letter_input + + pynini.closure(letter_input | NEMO_DIGIT) + + NEMO_DIGIT + + pynini.closure(letter_input | NEMO_DIGIT) + ) + currency_prefix_input = pynini.project( + pynini.string_file(get_abs_path("data/money/currency_prefix.tsv")), "input" + ) + currency_code_number = currency_prefix_input + pynini.closure(NEMO_DIGIT, 1) + raw_letter_leading_alnum = pynini.difference(raw_letter_leading_alnum, currency_code_number) + raw_short_digit_leading_alnum = NEMO_DIGIT + (pynini.accep("x") | pynini.accep("X")) + raw_mixed_alnum = raw_letter_leading_alnum | raw_short_digit_leading_alnum + alnum_reader = pynini.closure(letters | digit | pynini.accep(" "), 1) + alnum = (raw_mixed_alnum @ alnum_spacing @ alnum_reader).optimize() + + delimiter = insert_space + pynini.string_file(get_abs_path("data/serial/delimiter.tsv")) + insert_space + unit_input = pynini.project(pynini.string_file(get_abs_path("data/measure/unit.tsv")), "input") + numeric_measure_segment = pynini.closure(NEMO_DIGIT, 1) + unit_input + raw_alnum_segment = pynini.difference(raw_alnum, numeric_measure_segment) + raw_alnum_with_letter = ( + pynini.closure(letter_input | NEMO_DIGIT) + letter_input + pynini.closure(letter_input | NEMO_DIGIT) + ) + raw_alnum_with_letter = pynini.difference(raw_alnum_with_letter, numeric_measure_segment) + segment = (raw_alnum_segment @ alnum_spacing @ alnum_reader).optimize() + segment_with_letter = (raw_alnum_with_letter @ alnum_spacing @ alnum_reader).optimize() + delimited = segment_with_letter + pynini.closure( + delimiter + segment, 1 + ) | segment + delimiter + segment_with_letter + pynini.closure(delimiter + segment) + + special_word = pynini.string_file(get_abs_path("data/serial/words.tsv")) + covid_style = special_word + pynutil.delete("-") + insert_space + (NEMO_DIGIT**2 @ cardinal.just_cardinals) + + model_cue = pynini.string_file(get_abs_path("data/serial/model_cues.tsv")) + model_number = model_cue + insert_space + delimited + + graph = pynutil.add_weight(covid_style, -0.1) | model_number | delimited | alnum + self.fst = (pynutil.insert('name: "') + graph.optimize() + pynutil.insert('"')).optimize() diff --git a/nemo_text_processing/text_normalization/ja/taggers/telephone.py b/nemo_text_processing/text_normalization/ja/taggers/telephone.py new file mode 100644 index 000000000..744083686 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/taggers/telephone.py @@ -0,0 +1,125 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.text_normalization.ja.graph_utils import ( + NEMO_WHITE_SPACE, + GraphFst, + delete_space, + insert_space, +) +from nemo_text_processing.text_normalization.ja.utils import get_abs_path + + +class TelephoneFst(GraphFst): + """ + Finite state transducer for classifying Japanese telephone numbers. + + Examples: + 090-1234-5678 -> telephone { number_part: "ゼロ九ゼロ 一二三四 五六七八" preserve_order: true } + 03-1234-5678 -> telephone { number_part: "ゼロ三 一二三四 五六七八" preserve_order: true } + +81 90-1234-5678 -> telephone { country_code: "八一" number_part: "九ゼロ 一二三四 五六七八" preserve_order: true } + + Args: + deterministic: if True will provide a single transduction option + """ + + def __init__(self, deterministic: bool = True): + super().__init__(name="telephone", kind="classify", deterministic=deterministic) + + graph_digit = pynini.string_file(get_abs_path("data/numbers/digit.tsv")) + graph_zero = pynini.string_file(get_abs_path("data/numbers/zero.tsv")) + digit = graph_digit | graph_zero + extension_cue = pynini.project( + pynini.string_file(get_abs_path("data/telephone/extension.tsv")), + "input", + ) + + sep_char = pynini.union("-", "-", "ー", ".", ".") + delete_sep = pynutil.delete(sep_char) + delete_required_space = pynutil.delete(pynini.closure(NEMO_WHITE_SPACE, 1)) + block_sep = delete_space + (delete_sep | delete_required_space) + delete_space + insert_space + optional_after_paren_sep = delete_space + pynini.closure(delete_sep + delete_space, 0, 1) + required_country_sep = delete_space + (delete_sep | delete_required_space) + delete_space + + open_paren = pynutil.delete("(") | pynutil.delete("(") + close_paren = pynutil.delete(")") | pynutil.delete(")") + + digits = {count: digit**count for count in range(1, 11)} + paren = { + count: open_paren + digits[count] + close_paren + optional_after_paren_sep + insert_space + for count in range(1, 4) + } + + local_grouped_number = ( + digits[2] + block_sep + digits[4] + block_sep + digits[4] + | digits[3] + block_sep + digits[3] + block_sep + digits[4] + | digits[3] + block_sep + digits[4] + block_sep + digits[4] + | digits[4] + block_sep + digits[2] + block_sep + digits[4] + | digits[4] + block_sep + digits[3] + block_sep + digits[3] + | digits[4] + block_sep + digits[3] + block_sep + digits[4] + | digits[4] + block_sep + digits[4] + block_sep + digits[3] + ) + + local_parenthesized_number = ( + paren[2] + digits[4] + block_sep + digits[4] + | paren[3] + digits[3] + block_sep + digits[4] + | paren[3] + digits[4] + block_sep + digits[4] + ) + + compact_local_number = graph_zero + (digits[9] | digits[10]) + local_number = local_grouped_number | local_parenthesized_number | compact_local_number + + international_number = ( + digits[1] + block_sep + digits[4] + block_sep + digits[4] + | digits[2] + block_sep + digits[3] + block_sep + digits[4] + | digits[2] + block_sep + digits[4] + block_sep + digits[4] + | digits[3] + block_sep + digits[2] + block_sep + digits[4] + | digits[3] + block_sep + digits[3] + block_sep + digits[4] + | paren[1] + digits[4] + block_sep + digits[4] + | paren[2] + digits[3] + block_sep + digits[4] + | paren[2] + digits[4] + block_sep + digits[4] + | paren[3] + digits[2] + block_sep + digits[4] + | paren[3] + digits[3] + block_sep + digits[4] + ) + + country_code = digits[1] | digits[2] + country_code_component = ( + (pynutil.delete("+") | pynutil.delete("+")) + + pynutil.insert('country_code: "') + + country_code + + pynutil.insert('"') + + required_country_sep + + pynutil.insert(" ") + ) + + extension = ( + delete_space + + pynutil.delete(extension_cue) + + delete_space + + pynutil.insert(' extension: "') + + pynini.closure(digit, 1, 4) + + pynutil.insert('"') + ) + + number_part = pynutil.insert('number_part: "') + local_number + pynutil.insert('"') + international_number_part = pynutil.insert('number_part: "') + international_number + pynutil.insert('"') + + graph = number_part | (country_code_component + international_number_part) + graph = graph + pynini.closure(extension, 0, 1) + graph = graph + pynutil.insert(" preserve_order: true") + + self.fst = self.add_tokens(graph.optimize()).optimize() diff --git a/nemo_text_processing/text_normalization/ja/taggers/time.py b/nemo_text_processing/text_normalization/ja/taggers/time.py index 7c74bc53e..81c929cde 100644 --- a/nemo_text_processing/text_normalization/ja/taggers/time.py +++ b/nemo_text_processing/text_normalization/ja/taggers/time.py @@ -17,7 +17,7 @@ from pynini.lib import pynutil from nemo_text_processing.text_normalization.ja.graph_utils import GraphFst -from nemo_text_processing.text_normalization.ja.utils import get_abs_path +from nemo_text_processing.text_normalization.ja.utils import get_abs_path, load_labels class TimeFst(GraphFst): @@ -36,32 +36,39 @@ def __init__(self, cardinal: GraphFst, deterministic: bool = True): graph_cardinal = cardinal.just_cardinals hour_clock = pynini.string_file(get_abs_path("data/time/hour.tsv")) - minute_clock = pynini.string_file(get_abs_path("data/time/minute.tsv")) - second_clock = pynini.string_file(get_abs_path("data/time/second.tsv")) + minute_second_clock = pynini.string_file(get_abs_path("data/time/minute_second.tsv")) division = pynini.string_file(get_abs_path("data/time/division.tsv")) + zero_decimal = pynini.string_file(get_abs_path("data/numbers/zero_decimal.tsv")) + decimal_point = pynini.string_file(get_abs_path("data/numbers/decimal_point.tsv")) + suffixes = dict(load_labels(get_abs_path("data/time/suffix.tsv"))) + hour_suffix = suffixes["hour"] + hour_variants = pynini.union(hour_suffix, suffixes["duration_hour"], suffixes["approximate_hour"]) + minute_suffix = suffixes["minute"] + minute_modifier = pynini.union(suffixes["past"], suffixes["approximate"]) + half = suffixes["half"] + second_suffix = suffixes["second"] - division_component = pynutil.insert("suffix: \"") + division + pynutil.insert("\"") + division_component = pynutil.insert('suffix: "') + division + pynutil.insert('"') + hour_number = pynutil.add_weight(zero_decimal, -0.1) | graph_cardinal hour_component = ( - pynutil.insert("hours: \"") - + (graph_cardinal | (graph_cardinal + pynini.cross(".", "点") + graph_cardinal)) - + (pynini.accep("時") | pynini.accep("時間") | pynini.accep("時頃")) - + pynutil.insert("\"") + pynutil.insert('hours: "') + + (hour_number | (graph_cardinal + decimal_point + graph_cardinal)) + + hour_variants + + pynutil.insert('"') ) - minute_component = pynutil.insert("minutes: \"") + ( - graph_cardinal | (graph_cardinal + pynini.cross(".", "点") + graph_cardinal) - ) + pynini.accep("分") + pynini.closure((pynini.accep("過ぎ") | pynini.accep("頃")), 0, 1) + pynutil.insert( - "\"" - ) | ( - pynutil.insert("minutes: \"") - + pynini.accep("半") - + pynini.closure((pynini.accep("過ぎ") | pynini.accep("頃")), 0, 1) - + pynutil.insert("\"") + minute_component = pynutil.insert('minutes: "') + ( + graph_cardinal | (graph_cardinal + decimal_point + graph_cardinal) + ) + pynini.accep(minute_suffix) + pynini.closure(minute_modifier, 0, 1) + pynutil.insert('"') | ( + pynutil.insert('minutes: "') + + pynini.accep(half) + + pynini.closure(minute_modifier, 0, 1) + + pynutil.insert('"') ) second_component = ( - pynutil.insert("seconds: \"") - + (graph_cardinal | (graph_cardinal + pynini.cross(".", "点") + graph_cardinal)) - + pynini.accep("秒") - + pynutil.insert("\"") + pynutil.insert('seconds: "') + + (graph_cardinal | (graph_cardinal + decimal_point + graph_cardinal)) + + pynini.accep(second_suffix) + + pynutil.insert('"') ) graph_individual_time = pynini.closure(division_component + pynutil.insert(" "), 0, 1) + ( @@ -75,28 +82,28 @@ def __init__(self, cardinal: GraphFst, deterministic: bool = True): colon = pynutil.delete(":") hour_clock_component = ( - pynutil.insert("hours: \"") + pynutil.insert('hours: "') + pynutil.delete("0").ques + hour_clock - + pynutil.insert("時") - + pynutil.insert("\"") + + pynutil.insert(hour_suffix) + + pynutil.insert('"') ) minute_clock_component = ( - pynutil.insert("minutes: \"") + pynutil.insert('minutes: "') + pynutil.delete("0").ques - + minute_clock - + pynutil.insert("分") - + pynutil.insert("\"") + + minute_second_clock + + pynutil.insert(minute_suffix) + + pynutil.insert('"') ) second_clock_component = ( - pynutil.insert("seconds: \"") + pynutil.insert('seconds: "') + pynutil.delete("0").ques - + second_clock - + pynutil.insert("秒") - + pynutil.insert("\"") + + minute_second_clock + + pynutil.insert(second_suffix) + + pynutil.insert('"') ) - graph_clock = ( + graph_clock_with_seconds = ( hour_clock_component + pynutil.insert(" ") + colon @@ -104,7 +111,10 @@ def __init__(self, cardinal: GraphFst, deterministic: bool = True): + pynutil.insert(" ") + colon + second_clock_component - ) | (hour_clock_component + pynutil.insert(" ") + colon + minute_clock_component) + ) + graph_clock_with_minutes = hour_clock_component + pynutil.insert(" ") + colon + minute_clock_component + graph_clock_without_minutes = hour_clock_component + colon + pynutil.delete("00") + graph_clock = graph_clock_with_seconds | graph_clock_with_minutes | graph_clock_without_minutes graph = graph_individual_time | graph_clock diff --git a/nemo_text_processing/text_normalization/ja/taggers/tokenize_and_classify.py b/nemo_text_processing/text_normalization/ja/taggers/tokenize_and_classify.py index f992e9b70..52eaeb4a6 100644 --- a/nemo_text_processing/text_normalization/ja/taggers/tokenize_and_classify.py +++ b/nemo_text_processing/text_normalization/ja/taggers/tokenize_and_classify.py @@ -18,16 +18,36 @@ import pynini from pynini.lib import pynutil -from nemo_text_processing.text_normalization.ja.graph_utils import GraphFst, generator_main +from nemo_text_processing.text_normalization.ja.graph_utils import ( + GraphFst, + delete_extra_space, + delete_space, + generator_main, +) +from nemo_text_processing.text_normalization.ja.taggers.address import AddressFst from nemo_text_processing.text_normalization.ja.taggers.cardinal import CardinalFst from nemo_text_processing.text_normalization.ja.taggers.date import DateFst from nemo_text_processing.text_normalization.ja.taggers.decimal import DecimalFst +from nemo_text_processing.text_normalization.ja.taggers.electronic import ElectronicFst from nemo_text_processing.text_normalization.ja.taggers.fraction import FractionFst +from nemo_text_processing.text_normalization.ja.taggers.measure import MeasureFst +from nemo_text_processing.text_normalization.ja.taggers.money import MoneyFst from nemo_text_processing.text_normalization.ja.taggers.ordinal import OrdinalFst from nemo_text_processing.text_normalization.ja.taggers.punctuation import PunctuationFst +from nemo_text_processing.text_normalization.ja.taggers.range import RangeFst +from nemo_text_processing.text_normalization.ja.taggers.roman import RomanFst +from nemo_text_processing.text_normalization.ja.taggers.serial import SerialFst +from nemo_text_processing.text_normalization.ja.taggers.telephone import TelephoneFst from nemo_text_processing.text_normalization.ja.taggers.time import TimeFst from nemo_text_processing.text_normalization.ja.taggers.whitelist import WhiteListFst from nemo_text_processing.text_normalization.ja.taggers.word import WordFst +from nemo_text_processing.text_normalization.ja.verbalizers.cardinal import CardinalFst as CardinalVerbalizer +from nemo_text_processing.text_normalization.ja.verbalizers.date import DateFst as DateVerbalizer +from nemo_text_processing.text_normalization.ja.verbalizers.decimal import DecimalFst as DecimalVerbalizer +from nemo_text_processing.text_normalization.ja.verbalizers.fraction import FractionFst as FractionVerbalizer +from nemo_text_processing.text_normalization.ja.verbalizers.measure import MeasureFst as MeasureVerbalizer +from nemo_text_processing.text_normalization.ja.verbalizers.money import MoneyFst as MoneyVerbalizer +from nemo_text_processing.text_normalization.ja.verbalizers.time import TimeFst as TimeVerbalizer class ClassifyFst(GraphFst): @@ -59,7 +79,7 @@ def __init__( if cache_dir is not None and cache_dir != "None": os.makedirs(cache_dir, exist_ok=True) whitelist_file = os.path.basename(whitelist) if whitelist else "" - far_file = os.path.join(cache_dir, f"zh_tn_{deterministic}_deterministic_{whitelist_file}_tokenize.far") + far_file = os.path.join(cache_dir, f"ja_tn_{deterministic}_deterministic_{whitelist_file}_tokenize.far") if not overwrite_cache and far_file and os.path.exists(far_file): self.fst = pynini.Far(far_file, mode="r")["tokenize_and_classify"] else: @@ -68,25 +88,66 @@ def __init__( decimal = DecimalFst(cardinal=cardinal, deterministic=deterministic) time = TimeFst(cardinal=cardinal, deterministic=deterministic) fraction = FractionFst(cardinal=cardinal, deterministic=deterministic) + measure = MeasureFst(cardinal=cardinal, decimal=decimal, fraction=fraction, deterministic=deterministic) + money = MoneyFst(cardinal=cardinal, deterministic=deterministic) + telephone = TelephoneFst(deterministic=deterministic) ordinal = OrdinalFst(cardinal=cardinal, deterministic=deterministic) + address = AddressFst(cardinal=cardinal, deterministic=deterministic) + electronic = ElectronicFst(cardinal=cardinal, deterministic=deterministic) + roman = RomanFst(cardinal=cardinal, deterministic=deterministic) + serial = SerialFst(cardinal=cardinal, deterministic=deterministic) + + cardinal_verbalizer = CardinalVerbalizer(deterministic=deterministic) + decimal_verbalizer = DecimalVerbalizer(deterministic=deterministic) + fraction_verbalizer = FractionVerbalizer(deterministic=deterministic) + date_final = date.fst @ DateVerbalizer(deterministic=deterministic).fst + time_final = time.fst @ TimeVerbalizer(deterministic=deterministic).fst + money_final = money.fst @ MoneyVerbalizer(decimal=decimal_verbalizer, deterministic=deterministic).fst + measure_final = ( + measure.fst + @ MeasureVerbalizer( + cardinal=cardinal_verbalizer, + decimal=decimal_verbalizer, + fraction=fraction_verbalizer, + deterministic=deterministic, + ).fst + ) + range_graph = RangeFst( + cardinal=cardinal.fst @ cardinal_verbalizer.fst, + date=date_final, + time=time_final, + money=money_final, + measure=measure_final, + deterministic=deterministic, + ) whitelist = WhiteListFst(deterministic=deterministic) word = WordFst(deterministic=deterministic) punctuation = PunctuationFst(deterministic=deterministic) classify = pynini.union( + pynutil.add_weight(electronic.fst, 1.1), + pynutil.add_weight(address.fst, 1.1), + pynutil.add_weight(roman.fst, 1.1), + pynutil.add_weight(serial.fst, 1.1), + pynutil.add_weight(range_graph.fst, 1.1), pynutil.add_weight(date.fst, 1.1), - pynutil.add_weight(fraction.fst, 1.0), + pynutil.add_weight(fraction.fst, 1.1), + pynutil.add_weight(money.fst, 1.1), + pynutil.add_weight(measure.fst, 1.1), + pynutil.add_weight(telephone.fst, 1.1), pynutil.add_weight(time.fst, 1.1), pynutil.add_weight(whitelist.fst, 1.1), pynutil.add_weight(cardinal.fst, 1.1), - pynutil.add_weight(decimal.fst, 3.05), + pynutil.add_weight(decimal.fst, 1.1), pynutil.add_weight(ordinal.fst, 1.1), - pynutil.add_weight(punctuation.fst, 1.0), + pynutil.add_weight(punctuation.fst, 1.1), pynutil.add_weight(word.fst, 100), ) - token = pynutil.insert("tokens { ") + classify + pynutil.insert(" } ") - tagger = pynini.closure(token, 1) + token = pynutil.insert("tokens { ") + classify + pynutil.insert(" }") + tagger = ( + delete_space + token + pynini.closure((delete_extra_space | pynini.accep("")) + token) + delete_space + ) self.fst = tagger diff --git a/nemo_text_processing/text_normalization/ja/taggers/whitelist.py b/nemo_text_processing/text_normalization/ja/taggers/whitelist.py index 2f9391ddf..b272a8e22 100644 --- a/nemo_text_processing/text_normalization/ja/taggers/whitelist.py +++ b/nemo_text_processing/text_normalization/ja/taggers/whitelist.py @@ -17,7 +17,7 @@ import pynini from pynini.lib import pynutil -from nemo_text_processing.text_normalization.ja.graph_utils import GraphFst +from nemo_text_processing.text_normalization.ja.graph_utils import GraphFst, delete_space from nemo_text_processing.text_normalization.ja.utils import get_abs_path @@ -34,7 +34,9 @@ class WhiteListFst(GraphFst): def __init__(self, deterministic: bool = True): super().__init__(name="whitelist", kind="classify", deterministic=deterministic) - whitelist = pynini.string_file(get_abs_path("data/whitelist.tsv")) - graph = (pynutil.insert('name: "')) + (whitelist) + pynutil.insert('"') - - self.fst = graph.optimize() + whitelist = pynini.string_file(get_abs_path("data/whitelist.tsv")) + title = pynini.string_file(get_abs_path("data/whitelist_title.tsv")) + title_with_space = title + delete_space + pynutil.insert(" ") + graph = (pynutil.insert('name: "')) + (title_with_space | whitelist) + pynutil.insert('"') + + self.fst = graph.optimize() diff --git a/nemo_text_processing/text_normalization/ja/taggers/word.py b/nemo_text_processing/text_normalization/ja/taggers/word.py index b1403221b..7caaa0e37 100644 --- a/nemo_text_processing/text_normalization/ja/taggers/word.py +++ b/nemo_text_processing/text_normalization/ja/taggers/word.py @@ -1,30 +1,30 @@ -# Copyright (c) 2024 NVIDIA CORPORATION. All rights reserved. -# Copyright 2015 and onwards Google, Inc. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -from pynini.lib import pynutil - -from nemo_text_processing.text_normalization.ja.graph_utils import NEMO_NOT_SPACE, GraphFst - - -class WordFst(GraphFst): - """ - Finite state transducer for classifying plain tokens, that do not belong to any special class. This can be considered as the default class. - e.g. 文字 -> tokens { name: "文字" } - """ - - def __init__(self, deterministic: bool = True): - super().__init__(name="word", kind="classify", deterministic=deterministic) - word = pynutil.insert("name: \"") + NEMO_NOT_SPACE + pynutil.insert("\"") - self.fst = word.optimize() +# Copyright (c) 2024 NVIDIA CORPORATION. All rights reserved. +# Copyright 2015 and onwards Google, Inc. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +from pynini.lib import pynutil + +from nemo_text_processing.text_normalization.ja.graph_utils import NEMO_NOT_SPACE, GraphFst + + +class WordFst(GraphFst): + """ + Finite state transducer for classifying plain tokens, that do not belong to any special class. This can be considered as the default class. + e.g. 文字 -> tokens { name: "文字" } + """ + + def __init__(self, deterministic: bool = True): + super().__init__(name="word", kind="classify", deterministic=deterministic) + word = pynutil.insert("name: \"") + NEMO_NOT_SPACE + pynutil.insert("\"") + self.fst = word.optimize() diff --git a/nemo_text_processing/text_normalization/ja/verbalizers/decimal.py b/nemo_text_processing/text_normalization/ja/verbalizers/decimal.py index f4200001c..dbb28c9af 100644 --- a/nemo_text_processing/text_normalization/ja/verbalizers/decimal.py +++ b/nemo_text_processing/text_normalization/ja/verbalizers/decimal.py @@ -17,6 +17,7 @@ from pynini.lib import pynutil from nemo_text_processing.text_normalization.ja.graph_utils import NEMO_NOT_QUOTE, GraphFst, delete_space +from nemo_text_processing.text_normalization.ja.utils import get_abs_path, load_labels class DecimalFst(GraphFst): @@ -29,20 +30,22 @@ class DecimalFst(GraphFst): def __init__(self, deterministic: bool = True): super().__init__(name="decimal", kind="verbalize", deterministic=deterministic) - graph_integer = pynutil.delete("integer_part: \"") + pynini.closure(NEMO_NOT_QUOTE, 1) + pynutil.delete("\"") + decimal_point = load_labels(get_abs_path("data/numbers/decimal_point.tsv"))[0][1] + + graph_integer = pynutil.delete('integer_part: "') + pynini.closure(NEMO_NOT_QUOTE, 1) + pynutil.delete('"') graph_fraction = ( - pynutil.delete("fractional_part: \"") - + pynutil.insert("点") + pynutil.delete('fractional_part: "') + + pynutil.insert(decimal_point) + pynini.closure(NEMO_NOT_QUOTE, 1) - + pynutil.delete("\"") + + pynutil.delete('"') ) graph_optional_sign = pynini.closure( pynutil.delete("negative:") + delete_space - + pynutil.delete("\"") + + pynutil.delete('"') + pynini.closure(NEMO_NOT_QUOTE, 1) - + pynutil.delete("\"") + + pynutil.delete('"') ) graph_decimal_no_sign = graph_integer + pynutil.delete(" ") + graph_fraction diff --git a/nemo_text_processing/text_normalization/ja/verbalizers/electronic.py b/nemo_text_processing/text_normalization/ja/verbalizers/electronic.py new file mode 100644 index 000000000..1789062c9 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/verbalizers/electronic.py @@ -0,0 +1,147 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.text_normalization.ja.graph_utils import ( + NEMO_ALPHA, + NEMO_DIGIT, + NEMO_NARROW_NON_BREAK_SPACE, + NEMO_NOT_QUOTE, + NEMO_SIGMA, + GraphFst, + delete_preserve_order, + delete_space, + insert_space, +) +from nemo_text_processing.text_normalization.ja.utils import get_abs_path + + +class ElectronicFst(GraphFst): + """Verbalizes structured Japanese electronic tokens.""" + + def __init__(self, deterministic: bool = True): + super().__init__(name="electronic", kind="verbalize", deterministic=deterministic) + + symbol = pynini.string_file(get_abs_path("data/electronic/symbol.tsv")) + + def spaced_symbol(written: str): + return insert_space + (pynini.accep(written) @ symbol) + insert_space + + def insert_spoken_symbol(written: str): + return insert_space + (pynutil.insert(written) @ symbol) + insert_space + + dot = spaced_symbol(".") + hyphen = spaced_symbol("-") + slash = spaced_symbol("/") + insert_at = insert_spoken_symbol("@") + insert_colon = insert_spoken_symbol(":") + insert_slash = insert_spoken_symbol("/") + + digit = pynini.string_file(get_abs_path("data/numbers/digit.tsv")) + digit_zero_maru = digit | pynini.string_file(get_abs_path("data/numbers/zero_maru.tsv")) + digit_zero_user = digit | pynini.string_file(get_abs_path("data/numbers/zero.tsv")) + special_digit_run = pynutil.add_weight( + pynini.string_file(get_abs_path("data/electronic/special_digit_runs.tsv")), + -0.1, + ) + + raw_label = pynini.closure(NEMO_ALPHA | NEMO_DIGIT, 1) + alpha_label = pynini.closure(NEMO_ALPHA, 1) + digit_label = pynini.closure(NEMO_DIGIT, 1) + + insert_alpha_digit_space = pynini.cdrewrite(pynutil.insert(" "), NEMO_ALPHA, NEMO_DIGIT, NEMO_SIGMA) + insert_digit_alpha_space = pynini.cdrewrite(pynutil.insert(" "), NEMO_DIGIT, NEMO_ALPHA, NEMO_SIGMA) + alnum_spacing = insert_alpha_digit_space @ insert_digit_alpha_space + username_reader = pynini.closure( + NEMO_ALPHA | special_digit_run | digit_zero_user | pynini.accep(" "), + 1, + ) + raw_mixed_alnum = ( + pynini.closure(NEMO_ALPHA | NEMO_DIGIT) + + NEMO_ALPHA + + pynini.closure(NEMO_ALPHA | NEMO_DIGIT) + + NEMO_DIGIT + + pynini.closure(NEMO_ALPHA | NEMO_DIGIT) + ) | ( + pynini.closure(NEMO_ALPHA | NEMO_DIGIT) + + NEMO_DIGIT + + pynini.closure(NEMO_ALPHA | NEMO_DIGIT) + + NEMO_ALPHA + + pynini.closure(NEMO_ALPHA | NEMO_DIGIT) + ) + username_alnum = (raw_mixed_alnum @ alnum_spacing @ username_reader).optimize() + username_segment = pynutil.add_weight(username_alnum, -0.1) | alpha_label | digit_label + username_value = username_segment + pynini.closure((dot | hyphen) + username_segment) + + domain_value = raw_label + pynini.closure(dot + raw_label) + dot + alpha_label + path_label = raw_label + pynini.closure(hyphen + raw_label) + path_value = slash + path_label + pynini.closure(slash + path_label) + + username_field = ( + pynutil.delete("username:") + delete_space + pynutil.delete('"') + username_value + pynutil.delete('"') + ) + domain_field = ( + pynutil.delete("domain:") + delete_space + pynutil.delete('"') + domain_value + pynutil.delete('"') + ) + protocol_value = pynini.string_file(get_abs_path("data/electronic/protocol.tsv")) + protocol_field = ( + pynutil.delete("protocol:") + + delete_space + + pynutil.delete('"') + + protocol_value + + pynutil.delete('"') + + insert_colon + + insert_slash + + insert_slash + ) + path_field = pynutil.delete("path:") + delete_space + pynutil.delete('"') + path_value + pynutil.delete('"') + + email = username_field + delete_space + insert_at + domain_field + domain = domain_field + url = protocol_field + delete_space + domain_field + pynini.closure(delete_space + path_field, 0, 1) + + card_digit = pynini.closure(digit_zero_maru, 1) + protected_space = pynutil.insert(NEMO_NARROW_NON_BREAK_SPACE) + card_number_value = card_digit + pynini.closure(pynutil.delete(" ") + protected_space + card_digit) + card_number_field = ( + pynutil.delete("domain:") + delete_space + pynutil.delete('"') + card_number_value + pynutil.delete('"') + ) + card_cue = pynini.string_file(get_abs_path("data/electronic/card_cues.tsv")) | ( + pynini.string_file(get_abs_path("data/electronic/card_digit_count_prefix.tsv")) + + pynini.project(digit, "output") + + pynini.string_file(get_abs_path("data/electronic/card_digit_count_suffix.tsv")) + ) + card_cue_field = ( + pynutil.delete("protocol:") + delete_space + pynutil.delete('"') + card_cue + pynutil.delete('"') + ) + card = pynini.closure(card_cue_field + delete_space, 0, 1) + card_number_field + + filename_stem = pynini.closure( + pynini.difference(NEMO_NOT_QUOTE, pynini.union(".", "/", "@")), + 1, + ) + filename = ( + pynutil.delete("domain:") + + delete_space + + pynutil.delete('"') + + filename_stem + + protected_space + + pynini.string_file(get_abs_path("data/electronic/file_extensions.tsv")) + + pynutil.delete('"') + ) + + graph = (email | url | domain | card | filename) + delete_preserve_order + self.fst = self.delete_tokens(graph.optimize()).optimize() diff --git a/nemo_text_processing/text_normalization/ja/verbalizers/fraction.py b/nemo_text_processing/text_normalization/ja/verbalizers/fraction.py index 4743d3bcd..d14325b30 100644 --- a/nemo_text_processing/text_normalization/ja/verbalizers/fraction.py +++ b/nemo_text_processing/text_normalization/ja/verbalizers/fraction.py @@ -17,6 +17,7 @@ from pynini.lib import pynutil from nemo_text_processing.text_normalization.ja.graph_utils import NEMO_NOT_QUOTE, NEMO_SPACE, GraphFst, delete_space +from nemo_text_processing.text_normalization.ja.utils import get_abs_path, load_labels class FractionFst(GraphFst): @@ -24,7 +25,7 @@ class FractionFst(GraphFst): Finite state transducer for verbalizing fractionss, e.g. tokens { fraction { denominator: "二" numerator: "一"} } -> 1/2 tokens { fraction { integer: "一" denominator: "四" numerator: "三" } } -> 1と3/4 - tokens { fraction { integer: "1" denominator: "4" numerator: "3" } } -> 一荷四分の三 + tokens { fraction { integer: "1" denominator: "4" numerator: "3" } } -> 一と四分の三 tokens { fraction { denominator: "√3" numerator: "1" } } -> ルート三分の一 tokens { fraction { denominator: "1.65" numerator: "50" } } -> 一点六五分の五十 tokens { fraction { denominator: "二" numerator: "一"} } -> マイナス1/2 @@ -33,35 +34,41 @@ class FractionFst(GraphFst): def __init__(self, deterministic: bool = True): super().__init__(name="fraction", kind="verbalize", deterministic=deterministic) + markers = dict(load_labels(get_abs_path("data/fraction/marker.tsv"))) + fraction_marker = markers["fraction"] + root_written = markers["root_written"] + root_spoken = markers["root_spoken"] + mixed = markers["mixed"] + denominator_component = ( - pynutil.delete('denominator: \"') + pynini.closure(NEMO_NOT_QUOTE - "√") + pynutil.delete("\"") + pynutil.delete('denominator: "') + pynini.closure(NEMO_NOT_QUOTE - root_written) + pynutil.delete('"') ) numerator_component = ( - pynutil.delete('numerator: \"') + pynini.closure(NEMO_NOT_QUOTE - "√") + pynutil.delete("\"") + pynutil.delete('numerator: "') + pynini.closure(NEMO_NOT_QUOTE - root_written) + pynutil.delete('"') ) # 1/3 graph_regular_fraction = ( - denominator_component + pynutil.delete(NEMO_SPACE) + pynutil.insert("分の") + numerator_component + denominator_component + pynutil.delete(NEMO_SPACE) + pynutil.insert(fraction_marker) + numerator_component ) denominator_component_root = ( - pynutil.delete('denominator: \"') - + pynini.cross("√", "ルート") - + pynini.closure(NEMO_NOT_QUOTE - "√") - + pynutil.delete("\"") + pynutil.delete('denominator: "') + + pynini.cross(root_written, root_spoken) + + pynini.closure(NEMO_NOT_QUOTE - root_written) + + pynutil.delete('"') ) numerator_component_root = ( - pynutil.delete('numerator: \"') - + pynini.cross("√", "ルート") - + pynini.closure(NEMO_NOT_QUOTE - "√") - + pynutil.delete("\"") + pynutil.delete('numerator: "') + + pynini.cross(root_written, root_spoken) + + pynini.closure(NEMO_NOT_QUOTE - root_written) + + pynutil.delete('"') ) # √3/1 graph_regular_fraction_root = ( (denominator_component_root | denominator_component) + pynutil.delete(NEMO_SPACE) - + pynutil.insert("分の") + + pynutil.insert(fraction_marker) + (numerator_component_root | numerator_component) ) @@ -69,9 +76,9 @@ def __init__(self, deterministic: bool = True): graph_regular_fraction_char = ( (denominator_component | denominator_component_root) + pynutil.delete(NEMO_SPACE) - + pynutil.delete("morphosyntactic_features: \"") + + pynutil.delete('morphosyntactic_features: "') + pynini.closure(NEMO_NOT_QUOTE) - + pynutil.delete("\"") + + pynutil.delete('"') + pynutil.delete(NEMO_SPACE) + (numerator_component | numerator_component_root) ) @@ -79,23 +86,21 @@ def __init__(self, deterministic: bool = True): graph_integer = ( pynutil.delete("integer_part:") + delete_space - + pynutil.delete("\"") - + pynini.closure(pynini.cross("√", "ルート"), 0, 1) - + pynini.closure( - NEMO_NOT_QUOTE - pynini.union("荷", "と", "√") - ) # had to remove these 3 items fron nemo_not _quote so the root is properly converted in a deterministic way. - + pynutil.insert("荷") - + pynutil.delete("\"") + + pynutil.delete('"') + + pynini.closure(pynini.cross(root_written, root_spoken), 0, 1) + + pynini.closure(NEMO_NOT_QUOTE - pynini.union(mixed, root_written)) + + pynutil.insert(mixed) + + pynutil.delete('"') ) graph_integer_with_char = ( pynutil.delete("integer_part:") + delete_space - + pynutil.delete("\"") - + pynini.closure(pynini.cross("√", "ルート"), 0, 1) - + pynini.closure(NEMO_NOT_QUOTE - pynini.union("荷", "と", "√")) - + (pynini.accep("と") | pynini.accep("荷")) - + pynutil.delete("\"") + + pynutil.delete('"') + + pynini.closure(pynini.cross(root_written, root_spoken), 0, 1) + + pynini.closure(NEMO_NOT_QUOTE - pynini.union(mixed, root_written)) + + pynini.accep(mixed) + + pynutil.delete('"') ) graph_regular_integer = ( @@ -107,9 +112,9 @@ def __init__(self, deterministic: bool = True): optional_sign = ( pynutil.delete("negative:") + delete_space - + pynutil.delete("\"") + + pynutil.delete('"') + pynini.closure(NEMO_NOT_QUOTE) - + pynutil.delete("\"") + + pynutil.delete('"') + delete_space ) diff --git a/nemo_text_processing/text_normalization/ja/verbalizers/measure.py b/nemo_text_processing/text_normalization/ja/verbalizers/measure.py new file mode 100644 index 000000000..12057afdc --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/verbalizers/measure.py @@ -0,0 +1,50 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.text_normalization.ja.graph_utils import ( + NEMO_NOT_QUOTE, + GraphFst, + delete_preserve_order, + delete_space, +) + + +class MeasureFst(GraphFst): + """ + Finite state transducer for verbalizing Japanese measure tokens. + + Examples: + measure { cardinal { integer: "五" } units: "キロ" preserve_order: true } -> 五キロ + measure { cardinal { integer: "時速六十" } units: "キロ" preserve_order: true } -> 時速六十キロ + measure { cardinal { integer: "秒速五十" } units: "メートル" preserve_order: true } -> 秒速五十メートル + """ + + def __init__( + self, + cardinal: GraphFst, + decimal: GraphFst, + fraction: GraphFst, + deterministic: bool = True, + ): + super().__init__(name="measure", kind="verbalize", deterministic=deterministic) + + unit = delete_space + pynutil.delete('units: "') + pynini.closure(NEMO_NOT_QUOTE, 1) + pynutil.delete('"') + number = cardinal.fst | decimal.fst | fraction.fst + + graph = number + unit + delete_preserve_order + + self.fst = self.delete_tokens(graph.optimize()).optimize() diff --git a/nemo_text_processing/text_normalization/ja/verbalizers/money.py b/nemo_text_processing/text_normalization/ja/verbalizers/money.py new file mode 100644 index 000000000..83b8349cc --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/verbalizers/money.py @@ -0,0 +1,64 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.text_normalization.ja.graph_utils import ( + NEMO_NOT_QUOTE, + GraphFst, + delete_preserve_order, + delete_space, +) + + +class MoneyFst(GraphFst): + """ + Finite state transducer for verbalizing Japanese money. + + Example: + money { integer_part: "百" currency_maj: "円" preserve_order: true } -> 百円 + """ + + def __init__(self, decimal: GraphFst, deterministic: bool = True): + super().__init__(name="money", kind="verbalize", deterministic=deterministic) + + field_value = pynini.closure(NEMO_NOT_QUOTE, 1) + + integer_part = pynutil.delete('integer_part: "') + field_value + pynutil.delete('"') + + quantity = pynini.closure( + delete_space + pynutil.delete('quantity: "') + field_value + pynutil.delete('"'), + 0, + 1, + ) + + currency_major = delete_space + pynutil.delete('currency_maj: "') + field_value + pynutil.delete('"') + + fractional_part = pynini.closure( + delete_space + + pynutil.delete('fractional_part: "') + + field_value + + pynutil.delete('"') + + delete_space + + pynutil.delete('currency_min: "') + + field_value + + pynutil.delete('"'), + 0, + 1, + ) + + graph = integer_part + quantity + currency_major + fractional_part + delete_preserve_order + + self.fst = self.delete_tokens(graph.optimize()).optimize() diff --git a/nemo_text_processing/text_normalization/ja/verbalizers/post_processing.py b/nemo_text_processing/text_normalization/ja/verbalizers/post_processing.py index 8b196dcaf..e260abcd4 100644 --- a/nemo_text_processing/text_normalization/ja/verbalizers/post_processing.py +++ b/nemo_text_processing/text_normalization/ja/verbalizers/post_processing.py @@ -16,20 +16,26 @@ import os import pynini +from pynini.lib import pynutil -from nemo_text_processing.text_normalization.en.graph_utils import ( +from nemo_text_processing.text_normalization.ja.graph_utils import ( + NEMO_ALPHA, + NEMO_DIGIT, + NEMO_NARROW_NON_BREAK_SPACE, + NEMO_NON_BREAKING_SPACE, NEMO_NOT_SPACE, NEMO_SIGMA, - delete_space, generator_main, ) +from nemo_text_processing.text_normalization.ja.utils import get_abs_path, load_labels from nemo_text_processing.utils.logging import logger class PostProcessingFst: """ Finite state transducer that post-processing an entire sentence after verbalization is complete, e.g. - removes extra spaces around punctuation marks " ( one hundred and twenty three ) " -> "(one hundred and twenty three)" + removes extra spaces around punctuation marks + " ( one hundred and twenty three ) " -> "(one hundred and twenty three)" Args: cache_dir: path to a dir with .far grammar file. Set to None to avoid using cache. @@ -41,73 +47,98 @@ def __init__(self, cache_dir: str = None, overwrite_cache: bool = False): far_file = None if cache_dir is not None and cache_dir != "None": os.makedirs(cache_dir, exist_ok=True) - far_file = os.path.join(cache_dir, "zh_tn_post_processing.far") + far_file = os.path.join(cache_dir, "ja_tn_post_processing.far") if not overwrite_cache and far_file and os.path.exists(far_file): self.fst = pynini.Far(far_file, mode="r")["post_process_graph"] logger.info(f'Post processing graph was restored from {far_file}.') else: - self.set_punct_dict() self.fst = self.get_punct_postprocess_graph() if far_file: generator_main(far_file, {"post_process_graph": self.fst}) - def set_punct_dict(self): - self.punct_marks = { - "'": [ - "'", - '´', - 'ʹ', - 'ʻ', - 'ʼ', - 'ʽ', - 'ʾ', - 'ˈ', - 'ˊ', - 'ˋ', - '˴', - 'ʹ', - '΄', - '՚', - '՝', - 'י', - '׳', - 'ߴ', - 'ߵ', - 'ᑊ', - 'ᛌ', - '᾽', - '᾿', - '`', - '´', - '῾', - '‘', - '’', - '‛', - '′', - '‵', - 'ꞌ', - ''', - '`', - '𖽑', - '𖽒', - ], - } - def get_punct_postprocess_graph(self): """ - Returns graph to post process punctuation marks. + Returns graph to post process Japanese TN output. - {``} quotes are converted to {"}. Note, if there are spaces around single quote {'}, they will be kept. - By default, a space is added after a punctuation mark, and spaces are removed before punctuation marks. + Japanese verbalizers need ordinary inter-token spaces removed, but some + classes intentionally use spaces internally. Protect those spaces as NBSP + before deleting remaining technical spaces, then restore them as regular + spaces in the final output. """ - remove_space_around_single_quote = pynini.cdrewrite( - delete_space, NEMO_NOT_SPACE, NEMO_NOT_SPACE, pynini.closure(NEMO_SIGMA) + protect_space = pynini.cross(" ", NEMO_NON_BREAKING_SPACE) + ascii_char = NEMO_ALPHA | NEMO_DIGIT + phone_digit = ( + pynini.project(pynini.string_file(get_abs_path("data/numbers/digit.tsv")), "output") + | pynini.project(pynini.string_file(get_abs_path("data/numbers/zero.tsv")), "output") + | pynini.project(pynini.string_file(get_abs_path("data/numbers/zero_decimal.tsv")), "output") + | pynini.project(pynini.string_file(get_abs_path("data/numbers/zero_maru.tsv")), "output") + ).optimize() + space_sensitive_tokens = ( + pynini.project(pynini.string_file(get_abs_path("data/electronic/symbol.tsv")), "output") + | pynini.project(pynini.string_file(get_abs_path("data/latin/letters.tsv")), "output") + | pynini.project(pynini.string_file(get_abs_path("data/serial/words.tsv")), "output") + ).optimize() + title_tokens = pynini.union( + *{spoken for _, spoken in load_labels(get_abs_path("data/whitelist_title.tsv"))} + ).optimize() + ten = dict(load_labels(get_abs_path("data/numbers/teen.tsv")))["10"] + japanese_number = phone_digit | pynini.accep(ten) + sentence_suffix = pynini.string_file(get_abs_path("data/post_processing/sentence_suffix.tsv")) + collapse_double_space = pynini.cdrewrite(pynini.cross(" ", " "), "", "", pynini.closure(NEMO_SIGMA)) + + protect_whitelist_internal_space = pynini.closure(NEMO_SIGMA) + for spoken in {spoken for _, spoken in load_labels(get_abs_path("data/whitelist.tsv")) if " " in spoken}: + parts = spoken.split() + for left, right in zip(parts, parts[1:]): + protect_whitelist_internal_space @= pynini.cdrewrite( + protect_space, left, right, pynini.closure(NEMO_SIGMA) + ) + delete_ascii_inner_space = pynini.cdrewrite( + pynutil.delete(" "), ascii_char, ascii_char, pynini.closure(NEMO_SIGMA) + ) + protect_ascii_word_space = pynini.cdrewrite( + protect_space, ascii_char**2, ascii_char**2, pynini.closure(NEMO_SIGMA) + ) + protect_ascii_before_desu = pynini.cdrewrite( + protect_space, ascii_char**2, sentence_suffix, pynini.closure(NEMO_SIGMA) + ) + protect_ascii_before_japanese_number = pynini.cdrewrite( + protect_space, ascii_char**2, japanese_number, pynini.closure(NEMO_SIGMA) + ) + protect_title_before_ascii = pynini.cdrewrite( + protect_space, title_tokens, ascii_char**2, pynini.closure(NEMO_SIGMA) + ) + protect_after_space_sensitive_token = pynini.cdrewrite( + protect_space, space_sensitive_tokens, "", pynini.closure(NEMO_SIGMA) + ) + protect_before_space_sensitive_token = pynini.cdrewrite( + protect_space, "", space_sensitive_tokens, pynini.closure(NEMO_SIGMA) + ) + delete_technical_space = pynini.cdrewrite( + pynutil.delete(" "), NEMO_NOT_SPACE, NEMO_NOT_SPACE, pynini.closure(NEMO_SIGMA) + ) + restore_protected_space = pynini.cdrewrite( + pynini.cross(pynini.union(NEMO_NON_BREAKING_SPACE, NEMO_NARROW_NON_BREAK_SPACE), " "), + "", + "", + pynini.closure(NEMO_SIGMA), ) - # this works if spaces in between (good) - # delete space between 2 NEMO_NOT_SPACE(left and right to the space) that are with in a content of NEMO_SIGMA - graph = remove_space_around_single_quote.optimize() + graph = ( + collapse_double_space + @ collapse_double_space + @ protect_whitelist_internal_space + @ protect_ascii_word_space + @ delete_ascii_inner_space + @ protect_ascii_before_desu + @ protect_ascii_before_japanese_number + @ protect_title_before_ascii + @ protect_after_space_sensitive_token + @ protect_before_space_sensitive_token + @ delete_technical_space + @ restore_protected_space + ).optimize() return graph diff --git a/nemo_text_processing/text_normalization/ja/verbalizers/postprocessor.py b/nemo_text_processing/text_normalization/ja/verbalizers/postprocessor.py index 3ff05fa57..cd4b6291a 100644 --- a/nemo_text_processing/text_normalization/ja/verbalizers/postprocessor.py +++ b/nemo_text_processing/text_normalization/ja/verbalizers/postprocessor.py @@ -14,26 +14,20 @@ import pynini -from pynini.lib import pynutil, utf8 +from pynini.lib import pynutil -from nemo_text_processing.text_normalization.ja.graph_utils import ( - NEMO_ALPHA, - NEMO_DIGIT, - NEMO_PUNCT, - NEMO_SIGMA, - NEMO_WHITE_SPACE, - GraphFst, -) -from nemo_text_processing.text_normalization.ja.utils import get_abs_path +from nemo_text_processing.text_normalization.ja.graph_utils import NEMO_SIGMA, TO_LOWER, TO_UPPER, GraphFst +from nemo_text_processing.text_normalization.ja.taggers.punctuation import PunctuationFst class PostProcessor(GraphFst): - ''' - Postprocessing of TN, now contains: - 1. punctuation removal - 2. letter case conversion - 3. oov tagger - ''' + """ + Optional postprocessing for Japanese TN. + + The default graph is an identity rewrite. Optional punctuation removal and + ASCII case conversion are kept generic; OOV tagging needs a Japanese-specific + character inventory and is intentionally not implemented here. + """ def __init__( self, @@ -44,38 +38,20 @@ def __init__( ): super().__init__(name="PostProcessor", kind="processor") - graph = pynini.cdrewrite('', '', '', NEMO_SIGMA) + if to_upper and to_lower: + raise ValueError("to_upper and to_lower cannot both be enabled.") + if tag_oov: + raise ValueError("tag_oov is not supported for Japanese TN without a Japanese charset inventory.") + + graph = pynini.cdrewrite("", "", "", NEMO_SIGMA) if remove_puncts: - remove_puncts_graph = pynutil.delete( - pynini.union(NEMO_PUNCT, pynini.string_file(get_abs_path('data/char/punctuations_zh.tsv'))) - ) + remove_puncts_graph = pynutil.delete(pynini.union(*PunctuationFst().punct_marks)) graph @= pynini.cdrewrite(remove_puncts_graph, "", "", NEMO_SIGMA).optimize() - if to_upper or to_lower: - if to_upper: - conv_cases_graph = pynini.inverse(pynini.string_file(get_abs_path('data/char/upper_to_lower.tsv'))) - else: - conv_cases_graph = pynini.string_file(get_abs_path('data/char/upper_to_lower.tsv')) - + if to_upper: + graph @= pynini.cdrewrite(TO_UPPER, "", "", NEMO_SIGMA).optimize() + elif to_lower: + conv_cases_graph = TO_LOWER graph @= pynini.cdrewrite(conv_cases_graph, "", "", NEMO_SIGMA).optimize() - if tag_oov: - zh_charset_std = pynini.string_file(get_abs_path("data/char/charset_national_standard_2013_8105.tsv")) - zh_charset_ext = pynini.string_file(get_abs_path("data/char/charset_extension.tsv")) - - zh_charset = ( - zh_charset_std | zh_charset_ext | pynini.string_file(get_abs_path("data/char/punctuations_zh.tsv")) - ) - en_charset = NEMO_DIGIT | NEMO_ALPHA | NEMO_PUNCT | NEMO_WHITE_SPACE - charset = zh_charset | en_charset - - with open(get_abs_path("data/char/oov_tags.tsv"), "r") as f: - tags = f.readline().strip().split('\t') - assert len(tags) == 2 - ltag, rtag = tags - - oov_charset = pynini.difference(utf8.VALID_UTF8_CHAR, charset) - tag_oov_graph = pynutil.insert(ltag) + oov_charset + pynutil.insert(rtag) - graph @= pynini.cdrewrite(tag_oov_graph, "", "", NEMO_SIGMA).optimize() - self.fst = graph.optimize() diff --git a/nemo_text_processing/text_normalization/ja/verbalizers/telephone.py b/nemo_text_processing/text_normalization/ja/verbalizers/telephone.py new file mode 100644 index 000000000..f1584a8c5 --- /dev/null +++ b/nemo_text_processing/text_normalization/ja/verbalizers/telephone.py @@ -0,0 +1,76 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.text_normalization.ja.graph_utils import ( + NEMO_NARROW_NON_BREAK_SPACE, + NEMO_NOT_QUOTE, + GraphFst, + delete_preserve_order, + delete_space, +) +from nemo_text_processing.text_normalization.ja.utils import get_abs_path + + +class TelephoneFst(GraphFst): + """ + Finite state transducer for verbalizing Japanese telephone numbers. + + Example: + telephone { number_part: "ゼロ九ゼロ 一二三四 五六七八" preserve_order: true } + -> ゼロ九ゼロ、 一二三四、 五六七八 + """ + + def __init__(self, deterministic: bool = True): + super().__init__(name="telephone", kind="verbalize", deterministic=deterministic) + + country_code_prefix = pynini.string_file(get_abs_path("data/telephone/country_code_prefix.tsv")) + group_separator = pynini.string_file(get_abs_path("data/telephone/group_separator.tsv")) + extension_cue = pynini.project( + pynini.string_file(get_abs_path("data/telephone/extension.tsv")), + "output", + ) + country_code = ( + pynutil.delete('country_code: "') + + pynutil.insert(country_code_prefix) + + pynini.closure(NEMO_NOT_QUOTE, 1) + + pynutil.delete('"') + ) + + protected_space = pynutil.insert(NEMO_NARROW_NON_BREAK_SPACE) + spoken_group_separator = pynutil.insert(group_separator) + protected_space + number_group = pynini.closure( + (NEMO_NOT_QUOTE - " ") | (pynutil.delete(" ") + spoken_group_separator), + 1, + ) + number_part = pynutil.delete('number_part: "') + number_group + pynutil.delete('"') + + extension = ( + pynutil.delete('extension: "') + + pynutil.insert(extension_cue) + + protected_space + + pynini.closure(NEMO_NOT_QUOTE, 1) + + pynutil.delete('"') + ) + optional_extension = pynini.closure(delete_space + spoken_group_separator + extension, 0, 1) + + graph = ( + ((country_code + delete_space + spoken_group_separator + number_part) | number_part) + + optional_extension + + delete_preserve_order + ) + + self.fst = self.delete_tokens(graph).optimize() diff --git a/nemo_text_processing/text_normalization/ja/verbalizers/verbalize.py b/nemo_text_processing/text_normalization/ja/verbalizers/verbalize.py index 6a16f96d9..392f9fdde 100644 --- a/nemo_text_processing/text_normalization/ja/verbalizers/verbalize.py +++ b/nemo_text_processing/text_normalization/ja/verbalizers/verbalize.py @@ -19,8 +19,12 @@ from nemo_text_processing.text_normalization.ja.verbalizers.cardinal import CardinalFst from nemo_text_processing.text_normalization.ja.verbalizers.date import DateFst from nemo_text_processing.text_normalization.ja.verbalizers.decimal import DecimalFst +from nemo_text_processing.text_normalization.ja.verbalizers.electronic import ElectronicFst from nemo_text_processing.text_normalization.ja.verbalizers.fraction import FractionFst +from nemo_text_processing.text_normalization.ja.verbalizers.measure import MeasureFst +from nemo_text_processing.text_normalization.ja.verbalizers.money import MoneyFst from nemo_text_processing.text_normalization.ja.verbalizers.ordinal import OrdinalFst +from nemo_text_processing.text_normalization.ja.verbalizers.telephone import TelephoneFst from nemo_text_processing.text_normalization.ja.verbalizers.time import TimeFst from nemo_text_processing.text_normalization.ja.verbalizers.whitelist import WhiteListFst from nemo_text_processing.text_normalization.ja.verbalizers.word import WordFst @@ -45,11 +49,12 @@ def __init__(self, deterministic: bool = True): decimal = DecimalFst(deterministic=deterministic) word = WordFst(deterministic=deterministic) fraction = FractionFst(deterministic=deterministic) - - # money = MoneyFst(decimal=decimal, deterministic=deterministic) - # measure = MeasureFst(cardinal=cardinal, decimal=decimal, fraction=fraction, deterministic=deterministic) + money = MoneyFst(decimal=decimal, deterministic=deterministic) + measure = MeasureFst(cardinal=cardinal, decimal=decimal, fraction=fraction, deterministic=deterministic) + telephone = TelephoneFst(deterministic=deterministic) time = TimeFst(deterministic=deterministic) whitelist = WhiteListFst(deterministic=deterministic) + electronic = ElectronicFst(deterministic=deterministic) graph = pynini.union( date.fst, @@ -57,9 +62,13 @@ def __init__(self, deterministic: bool = True): ordinal.fst, decimal.fst, fraction.fst, + money.fst, + measure.fst, + telephone.fst, word.fst, time.fst, whitelist.fst, + electronic.fst, ) graph = pynini.closure(delete_space) + graph + pynini.closure(delete_space) diff --git a/nemo_text_processing/text_normalization/ja/verbalizers/verbalize_final.py b/nemo_text_processing/text_normalization/ja/verbalizers/verbalize_final.py index 750598649..82132de01 100644 --- a/nemo_text_processing/text_normalization/ja/verbalizers/verbalize_final.py +++ b/nemo_text_processing/text_normalization/ja/verbalizers/verbalize_final.py @@ -1,54 +1,56 @@ -# Copyright (c) 2024, NVIDIA CORPORATION & AFFILIATES. All rights reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - - -import os - -import pynini -from pynini.lib import pynutil - -from nemo_text_processing.text_normalization.ja.graph_utils import GraphFst, delete_space -from nemo_text_processing.text_normalization.ja.verbalizers.postprocessor import PostProcessor -from nemo_text_processing.text_normalization.ja.verbalizers.verbalize import VerbalizeFst - -# from nemo.utils import logging - - -class VerbalizeFinalFst(GraphFst): - """ """ - - def __init__(self, deterministic: bool = True, cache_dir: str = None, overwrite_cache: bool = False): - super().__init__(name="verbalize_final", kind="verbalize", deterministic=deterministic) - far_file = None - if cache_dir is not None and cache_dir != "None": - os.makedirs(cache_dir, exist_ok=True) - far_file = os.path.join(cache_dir, f"jp_tn_{deterministic}_deterministic_verbalizer.far") - if not overwrite_cache and far_file and os.path.exists(far_file): - self.fst = pynini.Far(far_file, mode="r")["verbalize"] - else: - token_graph = VerbalizeFst(deterministic=deterministic) - - token_verbalizer = ( - pynutil.delete("tokens {") + delete_space + token_graph.fst + delete_space + pynutil.delete(" }") - ) - verbalizer = pynini.closure(delete_space + token_verbalizer + delete_space) - - postprocessor = PostProcessor( - remove_puncts=False, - to_upper=False, - to_lower=False, - tag_oov=False, - ) - - self.fst = (verbalizer @ postprocessor.fst).optimize() +# Copyright (c) 2024, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + + +import os + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.text_normalization.ja.graph_utils import GraphFst, delete_space, generator_main +from nemo_text_processing.text_normalization.ja.verbalizers.postprocessor import PostProcessor +from nemo_text_processing.text_normalization.ja.verbalizers.verbalize import VerbalizeFst +from nemo_text_processing.utils.logging import logger + + +class VerbalizeFinalFst(GraphFst): + """ """ + + def __init__(self, deterministic: bool = True, cache_dir: str = None, overwrite_cache: bool = False): + super().__init__(name="verbalize_final", kind="verbalize", deterministic=deterministic) + far_file = None + if cache_dir is not None and cache_dir != "None": + os.makedirs(cache_dir, exist_ok=True) + far_file = os.path.join(cache_dir, f"ja_tn_{deterministic}_deterministic_verbalizer.far") + if not overwrite_cache and far_file and os.path.exists(far_file): + self.fst = pynini.Far(far_file, mode="r")["verbalize"] + logger.info(f"VerbalizeFinalFst graph was restored from {far_file}.") + else: + token_graph = VerbalizeFst(deterministic=deterministic) + + token_verbalizer = ( + pynutil.delete("tokens {") + delete_space + token_graph.fst + delete_space + pynutil.delete(" }") + ) + verbalizer = pynini.closure(delete_space + token_verbalizer + delete_space) + + postprocessor = PostProcessor( + remove_puncts=False, + to_upper=False, + to_lower=False, + tag_oov=False, + ) + + self.fst = (verbalizer @ postprocessor.fst).optimize() + if far_file: + generator_main(far_file, {"verbalize": self.fst}) diff --git a/nemo_text_processing/text_normalization/ja/verbalizers/whitelist.py b/nemo_text_processing/text_normalization/ja/verbalizers/whitelist.py index 11b0b3ae0..44d694323 100644 --- a/nemo_text_processing/text_normalization/ja/verbalizers/whitelist.py +++ b/nemo_text_processing/text_normalization/ja/verbalizers/whitelist.py @@ -16,7 +16,7 @@ import pynini from pynini.lib import pynutil -from nemo_text_processing.text_normalization.ja.graph_utils import NEMO_CHAR, NEMO_SIGMA, GraphFst, delete_space +from nemo_text_processing.text_normalization.ja.graph_utils import NEMO_NOT_QUOTE, NEMO_SIGMA, GraphFst, delete_space class WhiteListFst(GraphFst): @@ -28,11 +28,11 @@ class WhiteListFst(GraphFst): def __init__(self, deterministic: bool = True): super().__init__(name="whitelist", kind="verbalize", deterministic=deterministic) graph = ( - pynutil.delete("name:") - + delete_space - + pynutil.delete("\"") - + pynini.closure(NEMO_CHAR - " ", 1) - + pynutil.delete("\"") - ) + pynutil.delete("name:") + + delete_space + + pynutil.delete("\"") + + pynini.closure(NEMO_NOT_QUOTE, 1) + + pynutil.delete("\"") + ) graph = graph @ pynini.cdrewrite(pynini.cross(u"\u00a0", " "), "", "", NEMO_SIGMA) self.fst = graph.optimize() diff --git a/nemo_text_processing/text_normalization/ko/data/whitelist.tsv b/nemo_text_processing/text_normalization/ko/data/whitelist.tsv index 82dc1220e..d0bdf4caf 100644 --- a/nemo_text_processing/text_normalization/ko/data/whitelist.tsv +++ b/nemo_text_processing/text_normalization/ko/data/whitelist.tsv @@ -24,6 +24,7 @@ No. 번호 ) 오른쪽 괄호 + 더하기 - 마이너스 += 은 Σ 시그마 η 에타 κ 카파 diff --git a/nemo_text_processing/text_normalization/ko/taggers/cardinal.py b/nemo_text_processing/text_normalization/ko/taggers/cardinal.py index e2b0f87e3..49c36ec82 100644 --- a/nemo_text_processing/text_normalization/ko/taggers/cardinal.py +++ b/nemo_text_processing/text_normalization/ko/taggers/cardinal.py @@ -16,7 +16,7 @@ import pynini from pynini.lib import pynutil -from nemo_text_processing.text_normalization.ko.graph_utils import NEMO_DIGIT, NEMO_SPACE, GraphFst +from nemo_text_processing.text_normalization.ko.graph_utils import NEMO_DIGIT, NEMO_SPACE, GraphFst, delete_space from nemo_text_processing.text_normalization.ko.utils import get_abs_path @@ -274,6 +274,53 @@ def __init__(self, deterministic: bool = True): graph_zero, ).optimize() + # ---------------------------- + # Context-based digit-by-digit reading + # e.g., 번호는 0987654321 -> 번호는 영구팔칠육오사삼이일 + # + # Keep this separate from graph_num so regular cardinal zero + # and place-value behavior remain unchanged. + + serial_space = pynini.closure(delete_space) + + # Exclude 0 from graph_digit and force 0 -> 영. + # This avoids ambiguity if digit.tsv has another mapping for 0. + graph_digit_one_to_nine = (pynini.difference(NEMO_DIGIT, "0") @ graph_digit).optimize() + + serial_digit = pynini.union( + pynini.cross("0", "영"), + graph_digit_one_to_nine, + ).optimize() + + # Optional separators between individual digits. + serial_separator = pynini.union( + pynutil.delete("-"), + pynutil.delete("."), + pynutil.delete(" "), + ).optimize() + + # Require at least three digits. + serial_body = ( + (serial_digit + pynini.closure(serial_separator, 0, 1)) ** 2 + + serial_digit + + pynini.closure(pynini.closure(serial_separator, 0, 1) + serial_digit) + ).optimize() + + serial_signal = pynini.string_map( + [ + ("번호는", "번호는 "), + ("번호가", "번호가 "), + ("번호를", "번호를 "), + ("연락처는", "연락처는 "), + ("연락처가", "연락처가 "), + ("연락처를", "연락처를 "), + ] + ).optimize() + + self.serial = ( + pynutil.insert('integer: "') + serial_signal + serial_space + serial_body + pynutil.insert('"') + ).optimize() + # ---------------------------- # Native counting + counters # e.g., 3개, 2명, 10살 @@ -319,7 +366,7 @@ def __init__(self, deterministic: bool = True): signed_integer = (minus_prefix | plus_prefix).ques + integer_token # Prefer accounting-form first, then signed form - final_graph = paren_negative | signed_integer | counter_case + final_graph = self.serial | paren_negative | signed_integer | counter_case # Wrap with class tokens and finalize final_graph = self.add_tokens(final_graph) diff --git a/nemo_text_processing/text_normalization/ko/taggers/date.py b/nemo_text_processing/text_normalization/ko/taggers/date.py index 9748abc49..8b5d89d91 100644 --- a/nemo_text_processing/text_normalization/ko/taggers/date.py +++ b/nemo_text_processing/text_normalization/ko/taggers/date.py @@ -84,6 +84,8 @@ def __init__(self, cardinal: GraphFst, deterministic: bool = True): era = pynini.union("기원전", "기원후").optimize() signs = pynutil.delete("/") | pynutil.delete(".") | pynutil.delete("-") + date_sep = signs + pynini.closure(delete_space, 0, 1) + insert_space + # Strict digit ranges for M/D/Y and Y/M/D _d = pynini.union(*[pynini.accep(str(i)) for i in range(10)]) _1to9 = pynini.union(*[pynini.accep(str(i)) for i in range(1, 10)]) @@ -185,14 +187,24 @@ def __init__(self, cardinal: GraphFst, deterministic: bool = True): graph_basic_date = ( pynini.closure(era_component + insert_space, 0, 1) + year_component_y4_strict - + signs - + insert_space + + date_sep + (pynutil.insert("month: \"") + month_cardinal + pynutil.insert("월") + pynutil.insert("\"")) - + signs - + insert_space + + date_sep + (pynutil.insert("day: \"") + cardinal_lz + pynutil.insert("일") + pynutil.insert("\"")) - + pynini.closure(pynini.closure(insert_space, 0, 1) + week_component, 0, 1) + + pynini.closure(insert_space + week_component, 0, 1) ) + graph_basic_date_with_dot_weekday = ( + pynini.closure(era_component + insert_space, 0, 1) + + year_component_y4_strict + + date_sep + + (pynutil.insert("month: \"") + month_cardinal + pynutil.insert("월") + pynutil.insert("\"")) + + date_sep + + (pynutil.insert("day: \"") + cardinal_lz + pynutil.insert("일") + pynutil.insert("\"")) + + pynutil.delete(".") + + pynini.closure(delete_space, 0, 1) + + insert_space + + week_component + ).optimize() # American: MM/DD/YYYY graph_american_date = ( @@ -298,7 +310,8 @@ def __init__(self, cardinal: GraphFst, deterministic: bool = True): ).optimize() graph_all_date = ( - graph_basic_date + graph_basic_date_with_dot_weekday + | graph_basic_date | graph_american_date | graph_european_date | graph_individual_component diff --git a/nemo_text_processing/text_normalization/ko/taggers/telephone.py b/nemo_text_processing/text_normalization/ko/taggers/telephone.py index 90f31bb1f..e20cfc208 100644 --- a/nemo_text_processing/text_normalization/ko/taggers/telephone.py +++ b/nemo_text_processing/text_normalization/ko/taggers/telephone.py @@ -37,8 +37,9 @@ class TelephoneFst(GraphFst): def __init__(self, deterministic: bool = True): super().__init__(name="telephone", kind="classify", deterministic=deterministic) - # Separator between digit blocks (e.g., "-" or ".") - delete_sep = pynutil.delete("-") | pynutil.delete(".") + # Separator between number blocks. + delete_sep = pynutil.delete(pynini.union("-", ".", " ")).optimize() + # Optional space inserted between blocks insert_block_space = insert_space @@ -47,6 +48,7 @@ def __init__(self, deterministic: bool = True): zero_map = pynini.cross("0", "영") digit_ko = (digit | zero_map).optimize() + two_digits = digit_ko**2 three_digits = digit_ko**3 four_digits = digit_ko**4 @@ -62,25 +64,49 @@ def __init__(self, deterministic: bool = True): + delete_space ) - # area part: "123-" | "123." | "(123)" [space?] or "(123)-" - area_core = three_digits - area_part = ( - (area_core + delete_sep) - | ( - pynutil.delete("(") - + area_core - + pynutil.delete(")") - + pynini.closure(pynutil.delete(" "), 0, 1) - + pynini.closure(delete_sep, 0, 1) + # First block may contain 2 or 3 digits. + # Examples: 02, 031, 043, 010 + first_block = pynini.union( + two_digits, + three_digits, + ).optimize() + + # Middle block may contain 3 or 4 digits. + # Examples: 123, 1234 + middle_block = pynini.union( + three_digits, + four_digits, + ).optimize() + + # Plain telephone form: + # 02-1234-5678 + plain_first_part = (first_block + delete_sep + insert_block_space).optimize() + + # Parenthesized telephone form: + # (010)1234-5678 + parenthesized_first_part = ( + pynutil.delete("(") + + first_block + + pynutil.delete(")") + + pynini.closure( + pynutil.delete(pynini.union(" ", "-", ".")), + 0, + 1, ) - ) + insert_block_space + + insert_block_space + ).optimize() + + first_part = pynini.union( + plain_first_part, + parenthesized_first_part, + ).optimize() - # 2) allow 3 **or 4** digits in the middle block (to support 010-3713-7050) - mid = pynini.union(three_digits, four_digits) - last4 = four_digits + # Standard telephone layout: + # 2 or 3 digits + # followed by 3 or 4 digits + # followed by 4 digits + number_part_core = (first_part + middle_block + delete_sep + insert_block_space + four_digits).optimize() - # consume '-' or '.' between middle and last blocks - number_part_core = area_part + mid + delete_sep + insert_block_space + last4 number_part = pynutil.insert('number_part: "') + number_part_core + pynutil.insert('"') # final graph: with or without country code diff --git a/nemo_text_processing/text_normalization/ko/verbalizers/fraction.py b/nemo_text_processing/text_normalization/ko/verbalizers/fraction.py index 472b8a86d..d523dba00 100644 --- a/nemo_text_processing/text_normalization/ko/verbalizers/fraction.py +++ b/nemo_text_processing/text_normalization/ko/verbalizers/fraction.py @@ -129,43 +129,24 @@ def __init__(self, deterministic: bool = True): # Sigma for rewrite context (entire string) sigma = pynini.closure(NEMO_NOT_QUOTE | NEMO_SPACE) - # Fix subject particle agreement (이 → 가 for vowel-ending numerals) - # e.g., 사이 → 사가, 구이 → 구가 - subject_rewrite = pynini.cdrewrite( + # Fix particle agreement for vowel-ending numerals. + # Subject: 이 -> 가 + # Topic: 은 -> 는 + # Object: 을 -> 를 + particle_rewrite = pynini.cdrewrite( pynini.string_map( [ + # Subject particle ("이이", "이가"), ("사이", "사가"), ("오이", "오가"), ("구이", "구가"), - ] - ), - "", - "", - sigma, - ) - - # Fix topic particle agreement (은 → 는) - # e.g., 이은 → 이는, 사은 → 사는 - topic_rewrite = pynini.cdrewrite( - pynini.string_map( - [ + # Topic particle ("이은", "이는"), ("사은", "사는"), ("오은", "오는"), ("구은", "구는"), - ] - ), - "", - "", - sigma, - ) - - # Fix object particle agreement (을 → 를) - # e.g., 오을 → 오를, 이을 → 이를 - object_rewrite = pynini.cdrewrite( - pynini.string_map( - [ + # Object particle ("이을", "이를"), ("사을", "사를"), ("오을", "오를"), @@ -178,5 +159,5 @@ def __init__(self, deterministic: bool = True): ) # Apply all rewrite rules sequentially and final optimized FST - final_graph = final_graph @ subject_rewrite @ topic_rewrite @ object_rewrite + final_graph = final_graph @ particle_rewrite self.fst = final_graph.optimize() diff --git a/nemo_text_processing/text_normalization/normalize.py b/nemo_text_processing/text_normalization/normalize.py index d8ebf2f4d..c6a256eb5 100644 --- a/nemo_text_processing/text_normalization/normalize.py +++ b/nemo_text_processing/text_normalization/normalize.py @@ -177,7 +177,11 @@ def __init__( from nemo_text_processing.text_normalization.rw.verbalizers.verbalize_final import VerbalizeFinalFst elif lang == 'ja': from nemo_text_processing.text_normalization.ja.taggers.tokenize_and_classify import ClassifyFst + from nemo_text_processing.text_normalization.ja.verbalizers.post_processing import PostProcessingFst from nemo_text_processing.text_normalization.ja.verbalizers.verbalize_final import VerbalizeFinalFst + + if post_process: + self.post_processor = PostProcessingFst(cache_dir=cache_dir, overwrite_cache=overwrite_cache) elif lang == 'vi': from nemo_text_processing.text_normalization.vi.taggers.tokenize_and_classify import ClassifyFst from nemo_text_processing.text_normalization.vi.verbalizers.post_processing import PostProcessingFst @@ -391,7 +395,11 @@ def normalize( return text output = SPACE_DUP.sub(' ', output[1:]) - if self.lang in ["en", "hi", "vi"] and hasattr(self, 'post_processor') and self.post_processor is not None: + if ( + self.lang in ["en", "hi", "ja", "vi"] + and hasattr(self, 'post_processor') + and self.post_processor is not None + ): output = self.post_process(output) if punct_post_process: diff --git a/tests/nemo_text_processing/ar/test_sparrowhawk_normalization.sh b/tests/nemo_text_processing/ar/test_sparrowhawk_normalization.sh new file mode 100755 index 000000000..6998a6fbc --- /dev/null +++ b/tests/nemo_text_processing/ar/test_sparrowhawk_normalization.sh @@ -0,0 +1,71 @@ +#! /bin/sh +GRAMMARS_DIR=${1:-"/workspace/sparrowhawk/documentation/grammars"} +TEST_DIR=${2:-"/workspace/tests/ar"} + +runtest () { + input=$1 + echo "INPUT is $input" + cd ${GRAMMARS_DIR} + + while IFS= read -r testcase; do + IFS='~' read -r written spoken <<< "$testcase" + + escaped_written=$(printf '%s' "$written" | sed 's/\\/\\\\/g') + denorm_pred=$(echo "$escaped_written" | normalizer_main --config=sparrowhawk_configuration.ascii_proto 2>&1 | tail -n 1 | sed 's/\xC2\xA0/ /g') + + spoken="$(echo -e "${spoken}" | sed -e 's/^[[:space:]]*//' -e 's/[[:space:]]*$//')" + denorm_pred="$(echo -e "${denorm_pred}" | sed -e 's/^[[:space:]]*//' -e 's/[[:space:]]*$//')" + + assertEquals "$written" "$spoken" "$denorm_pred" + done < "$input" +} + +# For test files stored as expected~input (spoken~written). +runtest_swapped () { + input=$1 + echo "INPUT is $input" + cd ${GRAMMARS_DIR} + + while IFS= read -r testcase; do + IFS='~' read -r spoken written <<< "$testcase" + + escaped_written=$(printf '%s' "$written" | sed 's/\\/\\\\/g') + denorm_pred=$(echo "$escaped_written" | normalizer_main --config=sparrowhawk_configuration.ascii_proto 2>&1 | tail -n 1 | sed 's/\xC2\xA0/ /g') + + spoken="$(echo -e "${spoken}" | sed -e 's/^[[:space:]]*//' -e 's/[[:space:]]*$//')" + denorm_pred="$(echo -e "${denorm_pred}" | sed -e 's/^[[:space:]]*//' -e 's/[[:space:]]*$//')" + + assertEquals "$written" "$spoken" "$denorm_pred" + done < "$input" +} + +testTNCardinal() { + input=$TEST_DIR/data_text_normalization/test_cases_cardinal.txt + runtest $input +} + +testTNDecimal() { + input=$TEST_DIR/data_text_normalization/test_cases_decimal.txt + runtest $input +} + +testTNFraction() { + input=$TEST_DIR/data_text_normalization/test_cases_fraction.txt + runtest_swapped $input +} + +testTNMeasure() { + input=$TEST_DIR/data_text_normalization/test_cases_measure.txt + runtest_swapped $input +} + +testTNMoney() { + input=$TEST_DIR/data_text_normalization/test_cases_money.txt + runtest $input +} + +# Remove all command-line arguments +shift $# + +# Load shUnit2 +. /workspace/shunit2/shunit2 diff --git a/tests/nemo_text_processing/en/data_text_normalization/test_cases_ordinal.txt b/tests/nemo_text_processing/en/data_text_normalization/test_cases_ordinal.txt index 2e1b5ec7e..d4f073525 100644 --- a/tests/nemo_text_processing/en/data_text_normalization/test_cases_ordinal.txt +++ b/tests/nemo_text_processing/en/data_text_normalization/test_cases_ordinal.txt @@ -24,4 +24,4 @@ 21th~twenty one th 121st~one hundred twenty first 111th~one hundred eleventh -111st~one hundred eleven st \ No newline at end of file +111st~one one one st \ No newline at end of file diff --git a/tests/nemo_text_processing/en/data_text_normalization/test_cases_serial.txt b/tests/nemo_text_processing/en/data_text_normalization/test_cases_serial.txt index f0a6e0a3f..f142ceb7e 100644 --- a/tests/nemo_text_processing/en/data_text_normalization/test_cases_serial.txt +++ b/tests/nemo_text_processing/en/data_text_normalization/test_cases_serial.txt @@ -29,3 +29,5 @@ a 4-kilogram bag~a four-kilogram bag 100-car~one hundred-car 123/261788/2021~one hundred twenty three/two six one seven eight eight/two thousand twenty one 2*8~two asterisk eight +my pnr is t2000~my pnr is t two thousand +your otp is ab9453~your otp is ab nine four five three \ No newline at end of file diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_address.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_address.txt index 9989fa75c..d0554ce30 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_address.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_address.txt @@ -1,47 +1,44 @@ -700 ओक स्ट्रीट~सात शून्य शून्य ओक स्ट्रीट -११ जंगल रोड~एक एक जंगल रोड -301 पार्क एवेन्यू~तीन शून्य एक पार्क एवेन्यू -गली नंबर १७ जीएकगढ़~गली नंबर एक सात जीएकगढ़ -अदनान अपार्टमेंट फ्लैट नंबर 55~अदनान अपार्टमेंट फ्लैट नंबर पाँच पाँच +700 ओक स्ट्रीट~सात सौ ओक स्ट्रीट +११ जंगल रोड~ग्यारह जंगल रोड +301 पार्क एवेन्यू~तीन सौ एक पार्क एवेन्यू +गली नंबर १७ जीएकगढ़~गली नंबर सत्रह जीएकगढ़ +अदनान अपार्टमेंट फ्लैट नंबर 55~अदनान अपार्टमेंट फ्लैट नंबर पचपन प्लॉट नंबर ८ बालाजी मार्केट~प्लॉट नंबर आठ बालाजी मार्केट -शॉप नंबर 109 9 और 10 डिवाइडिंग रोड सेक्टर 10 फरीदाबाद~शॉप नंबर एक शून्य नौ नौ और एक शून्य डिवाइडिंग रोड सेक्टर एक शून्य फरीदाबाद -बूथ ७०, सेक्टर ८, चंडीगढ़~बूथ सात शून्य, सेक्टर आठ, चंडीगढ़ -2221 Southern Street~दो दो दो एक सदर्न स्ट्रीट -७०० ओक स्ट्रीट~सात शून्य शून्य ओक स्ट्रीट -625 स्कूल स्ट्रीट~छह दो पाँच स्कूल स्ट्रीट +शॉप नंबर 109 9 और 10 डिवाइडिंग रोड सेक्टर 10 फरीदाबाद~शॉप नंबर एक सौ नौ नौ और दस डिवाइडिंग रोड सेक्टर दस फरीदाबाद +बूथ ७०, सेक्टर ८, चंडीगढ़~बूथ सत्तर, सेक्टर आठ, चंडीगढ़ +७०० ओक स्ट्रीट~सात सौ ओक स्ट्रीट +625 स्कूल स्ट्रीट~छह सौ पच्चीस स्कूल स्ट्रीट १४७० एस वाशिंगटन स्ट्रीट~एक चार सात शून्य एस वाशिंगटन स्ट्रीट -506 स्टेट रोड~पाँच शून्य छह स्टेट रोड -६६-४ पार्कहर्स्ट आर डी~छह छह हाइफ़न चार पार्कहर्स्ट आर डी -579 ट्रॉय-शेंक्टाडी रोड~पाँच सात नौ ट्रॉय हाइफ़न शेंक्टाडी रोड +506 स्टेट रोड~पाँच सौ छह स्टेट रोड +579 ट्रॉय-शेंक्टाडी रोड~पाँच सौ उनासी ट्रॉय हाइफ़न शेंक्टाडी रोड ७८३० - ई वेटरन्स पार्कवे, कोलंबस, जी ए ३१९०९~सात आठ तीन शून्य हाइफ़न ई वेटरन्स पार्कवे, कोलंबस, जी ए तीन एक नौ शून्य नौ -66-4, पार्कहर्स्ट रोड~छह छह हाइफ़न चार, पार्कहर्स्ट रोड -८४०/१, १०० फीट रोड, मेट्रो पिलर ५६-५७, इंदिरानगर, बैंगलोर~आठ चार शून्य बटा एक, एक शून्य शून्य फीट रोड, मेट्रो पिलर पाँच छह हाइफ़न पाँच सात, इंदिरानगर, बैंगलोर -17-18, राजलक्ष्मी नगर, 7th क्रॉस स्ट्रीट, 100 फीट बाईपास रोड, वेलाचेरी, चेन्नई~एक सात हाइफ़न एक आठ, राजलक्ष्मी नगर, सेवंथ क्रॉस स्ट्रीट, एक शून्य शून्य फीट बाईपास रोड, वेलाचेरी, चेन्नई +66-4, पार्कहर्स्ट रोड~छियासठ हाइफ़न चार, पार्कहर्स्ट रोड +८४०/१, १०० फीट रोड, मेट्रो पिलर ५६-५७, इंदिरानगर, बैंगलोर~आठ सौ चालीस बटा एक, एक सौ फीट रोड, मेट्रो पिलर छप्पन हाइफ़न सत्तावन, इंदिरानगर, बैंगलोर +17-18, राजलक्ष्मी नगर, 7th क्रॉस स्ट्रीट, 100 फीट बाईपास रोड, वेलाचेरी, चेन्नई~सत्रह हाइफ़न अठारह, राजलक्ष्मी नगर, सेवंथ क्रॉस स्ट्रीट, एक सौ फीट बाईपास रोड, वेलाचेरी, चेन्नई ४/५ न्यू म्युनिसिपल मार्केट रोड नंबर ५ और ६ सेन्टाक्रूज़ वेस्ट~चार बटा पाँच न्यू म्युनिसिपल मार्केट रोड नंबर पाँच और छह सेन्टाक्रूज़ वेस्ट -16/17 4th फ्लोर जवाहर नगर मटरू मंदिर रोड नंबर 2~एक छह बटा एक सात फ़ोर्थ फ्लोर जवाहर नगर मटरू मंदिर रोड नंबर दो -५/३०४ सिक्का कॉम्प्लेक्स विकास मार्ग एक्सटेंशन~पाँच बटा तीन शून्य चार सिक्का कॉम्प्लेक्स विकास मार्ग एक्सटेंशन -21/2 2nd फ्लोर 1st मेन रोड गांधी नगर~दो एक बटा दो सेकंड फ्लोर फ़र्स्ट मेन रोड गांधी नगर -नंबर २२/१८ ३rd फ्लोर सराय बोउ अली शू मार्केट~नंबर दो दो बटा एक आठ थर्ड फ्लोर सराय बोउ अली शू मार्केट -14/3, मथुरा रोड~एक चार बटा तीन, मथुरा रोड -यूनिट ३ १st फ्लोर नंबर ३७ सोलेमान खतर स्ट्रीट~यूनिट तीन फ़र्स्ट फ्लोर नंबर तीन सात सोलेमान खतर स्ट्रीट -1st फ्लोर नंबर 52 नॉर्थ अबूज़र स्ट्रीट खान ए अंसारी स्ट्रीट शरीयती स्ट्रीट 16617~फ़र्स्ट फ्लोर नंबर पाँच दो नॉर्थ अबूज़र स्ट्रीट खान ए अंसारी स्ट्रीट शरीयती स्ट्रीट एक छह छह एक सात -२०६ जय कॉम कॉम्प्लेक्स १st पोखरन रोड~दो शून्य छह जय कॉम कॉम्प्लेक्स फ़र्स्ट पोखरन रोड -नंबर 36 2nd फ्लोर सुपर 8 फेज 1 एकबतन टाउन तेहरान 13947~नंबर तीन छह सेकंड फ्लोर सुपर आठ फेज एक एकबतन टाउन तेहरान एक तीन नौ चार सात -२nd फ्लोर नंबर ८०८ आजादी स्ट्रीट~सेकंड फ्लोर नंबर आठ शून्य आठ आजादी स्ट्रीट -2nd फ्लोर नंबर 15 बिफ़ोर कांदि स्ट्रीट नॉर्थ सोहरावर्दी स्ट्रीट 15669~सेकंड फ्लोर नंबर एक पाँच बिफ़ोर कांदि स्ट्रीट नॉर्थ सोहरावर्दी स्ट्रीट एक पाँच छह छह नौ -यूनिट ४ नंबर २५ २nd गोलहा स्ट्रीट काशनी स्ट्रीट नूर स्क्वेर~यूनिट चार नंबर दो पाँच सेकंड गोलहा स्ट्रीट काशनी स्ट्रीट नूर स्क्वेर -ईस्ट 3rd फ्लोर नंबर 70 नेक्स्ट दो तोहीद इंस्टीट्यूट परचम स्ट्रीट~ईस्ट थर्ड फ्लोर नंबर सात शून्य नेक्स्ट दो तोहीद इंस्टीट्यूट परचम स्ट्रीट +16/17 4th फ्लोर जवाहर नगर मटरू मंदिर रोड नंबर 2~सोलह बटा सत्रह फ़ोर्थ फ्लोर जवाहर नगर मटरू मंदिर रोड नंबर दो +५/३०४ सिक्का कॉम्प्लेक्स विकास मार्ग एक्सटेंशन~पाँच बटा तीन सौ चार सिक्का कॉम्प्लेक्स विकास मार्ग एक्सटेंशन +21/2 2nd फ्लोर 1st मेन रोड गांधी नगर~इक्कीस बटा दो सेकंड फ्लोर फ़र्स्ट मेन रोड गांधी नगर +नंबर २२/१८ ३rd फ्लोर सराय बोउ अली शू मार्केट~नंबर बाईस बटा अठारह थर्ड फ्लोर सराय बोउ अली शू मार्केट +14/3, मथुरा रोड~चौदह बटा तीन, मथुरा रोड +यूनिट ३ १st फ्लोर नंबर ३७ सोलेमान खतर स्ट्रीट~यूनिट तीन फ़र्स्ट फ्लोर नंबर सैंतीस सोलेमान खतर स्ट्रीट +1st फ्लोर नंबर 52 नॉर्थ अबूज़र स्ट्रीट खान ए अंसारी स्ट्रीट शरीयती स्ट्रीट 16617~फ़र्स्ट फ्लोर नंबर बावन नॉर्थ अबूज़र स्ट्रीट खान ए अंसारी स्ट्रीट शरीयती स्ट्रीट एक छह छह एक सात +२०६ जय कॉम कॉम्प्लेक्स १st पोखरन रोड~दो सौ छह जय कॉम कॉम्प्लेक्स फ़र्स्ट पोखरन रोड +नंबर 36 2nd फ्लोर सुपर 8 फेज 1 एकबतन टाउन तेहरान 13947~नंबर छत्तीस सेकंड फ्लोर सुपर आठ फेज एक एकबतन टाउन तेहरान एक तीन नौ चार सात +२nd फ्लोर नंबर ८०८ आजादी स्ट्रीट~सेकंड फ्लोर नंबर आठ सौ आठ आजादी स्ट्रीट +2nd फ्लोर नंबर 15 बिफ़ोर कांदि स्ट्रीट नॉर्थ सोहरावर्दी स्ट्रीट 15669~सेकंड फ्लोर नंबर पंद्रह बिफ़ोर कांदि स्ट्रीट नॉर्थ सोहरावर्दी स्ट्रीट एक पाँच छह छह नौ +यूनिट ४ नंबर २५ २nd गोलहा स्ट्रीट काशनी स्ट्रीट नूर स्क्वेर~यूनिट चार नंबर पच्चीस सेकंड गोलहा स्ट्रीट काशनी स्ट्रीट नूर स्क्वेर +ईस्ट 3rd फ्लोर नंबर 70 नेक्स्ट दो तोहीद इंस्टीट्यूट परचम स्ट्रीट~ईस्ट थर्ड फ्लोर नंबर सत्तर नेक्स्ट दो तोहीद इंस्टीट्यूट परचम स्ट्रीट ३rd फ्लोर नंबर ५ हमेदन एली अपोज़िट लाले पार्क नॉर्थ कारगर स्ट्रीट~थर्ड फ्लोर नंबर पाँच हमेदन एली अपोज़िट लाले पार्क नॉर्थ कारगर स्ट्रीट 4th फ्लोर नंबर 1124 जमहोरी स्ट्रीट~फ़ोर्थ फ्लोर नंबर एक एक दो चार जमहोरी स्ट्रीट ५th फ्लोर नंबर ७/१ १३th एली शाहिद अराबली स्ट्रीट~फ़िफ्थ फ्लोर नंबर सात बटा एक थर्टींथ एली शाहिद अराबली स्ट्रीट -11, 80 फीट रोड, इंडियन ऑयल पेट्रोल पंप, कोरमंगला 6th ब्लॉक, बैंगलोर के सामने~एक एक, आठ शून्य फीट रोड, इंडियन ऑयल पेट्रोल पंप, कोरमंगला सिक्स्थ ब्लॉक, बैंगलोर के सामने -२१/११, जे ब्लॉक, ६th एवेन्यू मेन रोड, अन्ना नगर पूर्व, चेन्नई~दो एक बटा एक एक, जे ब्लॉक, सिक्स्थ एवेन्यू मेन रोड, अन्ना नगर पूर्व, चेन्नई -32A नाज़ प्लाज़ा मेरिस रोड~तीन दो ए नाज़ प्लाज़ा मेरिस रोड -२१४ बी गोविंद पूरी स्ट्रीट नंबर २~दो एक चार बी गोविंद पूरी स्ट्रीट नंबर दो -4362 16वीं एवेन्यू एसडब्ल्यू, देवदार रैपिड्स, आई ए 52404~चार तीन छह दो सोलहवीं एवेन्यू एसडब्ल्यू, देवदार रैपिड्स, आई ए बावन हज़ार चार सौ चार +11, 80 फीट रोड, इंडियन ऑयल पेट्रोल पंप, कोरमंगला 6th ब्लॉक, बैंगलोर के सामने~ग्यारह, अस्सी फीट रोड, इंडियन ऑयल पेट्रोल पंप, कोरमंगला सिक्स्थ ब्लॉक, बैंगलोर के सामने +२१/११, जे ब्लॉक, ६th एवेन्यू मेन रोड, अन्ना नगर पूर्व, चेन्नई~इक्कीस बटा ग्यारह, जे ब्लॉक, सिक्स्थ एवेन्यू मेन रोड, अन्ना नगर पूर्व, चेन्नई +32A नाज़ प्लाज़ा मेरिस रोड~बत्तीस ए नाज़ प्लाज़ा मेरिस रोड +२१४ बी गोविंद पूरी स्ट्रीट नंबर २~दो सौ चौदह बी गोविंद पूरी स्ट्रीट नंबर दो +२५१३ ५३ एवेन्यू, मुंबई, महाराष्ट्र ४००००१~दो पाँच एक तीन तिरेपन एवेन्यू, मुंबई, महाराष्ट्र चार शून्य शून्य शून्य शून्य एक अमरावती ६५५९३०~अमरावती छह पाँच पाँच नौ तीन शून्य शिमला, हिमाचल प्रदेश 593988~शिमला, हिमाचल प्रदेश पाँच नौ तीन नौ आठ आठ -२७०४४० डॉसन आर डी, अल्बानी, जीए ३१७०७~दो सात शून्य चार चार शून्य डॉसन आर डी, अल्बानी, जीए तीन एक सात शून्य सात रांची, झारखंड 736557~रांची, झारखंड सात तीन छह पाँच पाँच सात कोहिमा, नागालैंड ४४८३७७~कोहिमा, नागालैंड चार चार आठ तीन सात सात मुंबई, महाराष्ट्र 839488~मुंबई, महाराष्ट्र आठ तीन नौ चार आठ आठ diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_cardinal.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_cardinal.txt index d607992d7..050310f9f 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_cardinal.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_cardinal.txt @@ -148,3 +148,14 @@ ०७३~शून्य सात तीन 0001~शून्य शून्य शून्य एक ०००~शून्य शून्य शून्य +3,24,50,000~तीन करोड़ चौबीस लाख पचास हज़ार +२,१२,१५,०००~दो करोड़ बारह लाख पंद्रह हज़ार +32,450,000~तीन करोड़ चौबीस लाख पचास हज़ार +४,९९,९९,०००~चार करोड़ निन्यानबे लाख निन्यानबे हज़ार +5,50,00,000~पाँच करोड़ पचास लाख +32,45,000~बत्तीस लाख पैंतालीस हज़ार +५,५६,३२०~पाँच लाख छप्पन हज़ार तीन सौ बीस +1,23,456~एक लाख तेईस हज़ार चार सौ छप्पन +12,345~बारह हज़ार तीन सौ पैंतालीस +११,२२०~ग्यारह हज़ार दो सौ बीस +1,00,00,000~एक करोड़ \ No newline at end of file diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_date.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_date.txt index 86f1f6678..ac12e6af6 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_date.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_date.txt @@ -1,20 +1,19 @@ -06-05~छः मई +06-05~छह मई ३१-०६~इकतीस जून 02-01~दो जनवरी ०४-०१~चार जनवरी -01-10~एक अक्टूबर +01-10~एक अक्टूबर १२-०७~बारह जुलाई -02-27~फ़रवरी सत्ताईस -०४-०३~चार मार्च +०४-०३~चार मार्च 25-03-2020~पच्चीस मार्च दो हज़ार बीस ३०-०५-२०७०~तीस मई दो हज़ार सत्तर -12-07-1970~बारह जुलाई उन्नीस सौ सत्तर ०९-१२-२१०१~नौ दिसंबर इक्कीस सौ एक 23-08-2024~तेईस अगस्त दो हज़ार चौबीस -१०-२९-२०००~अक्टूबर उनतीस दो हज़ार -11-14-1100~नवंबर चौदह ग्यारह सौ -०३-२०१०~मार्च दो हज़ार दस -11-2024~नवंबर दो हज़ार चौबीस +३ मार्च~तीन मार्च +६ मार्च, २०१०~छह मार्च दो हज़ार दस +३१ मई, १९९० ई.~इकतीस मई उन्नीस सौ नब्बे ईसवी +मार्च, २०२४~मार्च दो हज़ार चौबीस +जनवरी, १९९० ई.~जनवरी उन्नीस सौ नब्बे ईसवी २०७०~दो हज़ार सत्तर 2024~दो हज़ार चौबीस १२० ई. पू.~एक सौ बीस ईसा पूर्व @@ -31,4 +30,10 @@ सन 1999~सन उन्नीस सौ निन्यानबे सन् १९२०~सन् उन्नीस सौ बीस साल 1971~साल उन्नीस सौ इकहत्तर -१९२०-२६ तक~उन्नीस सौ बीस से छब्बीस तक \ No newline at end of file +सन 1999 में~सन उन्नीस सौ निन्यानबे में +सन् उन्नीस सौ बीस~सन् उन्नीस सौ बीस +सन उन्नीस सौ बीस में~सन उन्नीस सौ बीस में +१९२०-२६ तक~उन्नीस सौ बीस से छब्बीस तक +2-7-1970~दो जुलाई उन्नीस सौ सत्तर +02-07-1970~दो जुलाई उन्नीस सौ सत्तर +12-07-1970~बारह जुलाई उन्नीस सौ सत्तर \ No newline at end of file diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_decimal.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_decimal.txt index 3582aff50..03b01de2f 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_decimal.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_decimal.txt @@ -18,3 +18,7 @@ १०००००००००००००.०००३~एक नील दशमलव शून्य शून्य शून्य तीन 1000000000000000.008~एक पद्म दशमलव शून्य शून्य आठ १०००००००००००००००००.४१२~एक शंख दशमलव चार एक दो +१९२.१६८~एक सौ बानबे दशमलव एक छह आठ +192.168~एक सौ बानबे दशमलव एक छह आठ +99.99~निन्यानबे दशमलव नौ नौ +९९.९९~निन्यानबे दशमलव नौ नौ diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_electronic.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_electronic.txt index 85b34c4a3..3265724a3 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_electronic.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_electronic.txt @@ -1,50 +1,50 @@ -gmail.com~जीमेल डॉट कॉम -yahoo.com~याहू डॉट कॉम -hotmail.com~हॉटमेल डॉट कॉम -google.com~गूगल डॉट कॉम -kumaar.org~के यू एम ए ए आर डॉट ऑर्ग -kumaar.info~के यू एम ए ए आर डॉट इन्फो -kumar@gmail.com~के यू एम ए आर एट जीमेल डॉट कॉम -robin@hotmail.com~रॉबिन एट हॉटमेल डॉट कॉम -kapil@live.com~के ए पी आई एल एट लाइव डॉट कॉम -sneha@live.com~एस एन ई एच ए एट लाइव डॉट कॉम -mayank@google.com~एम ए वाई ए एन के एट गूगल डॉट कॉम -charu@yahoo.com~सी एच ए आर यू एट याहू डॉट कॉम -john20@yahoo.com~जे ओ एच एन दो शून्य एट याहू डॉट कॉम -vivaan62@gmail.com~वी आई वी ए ए एन छह दो एट जीमेल डॉट कॉम -viaan15@kumaar.com~वी आई ए ए एन एक पाँच एट के यू एम ए ए आर डॉट कॉम -ltaa12@gmail.com~एल टी ए ए एक दो एट जीमेल डॉट कॉम -kristen11@hotmail.com~के आर आई एस टी ई एन एक एक एट हॉटमेल डॉट कॉम -dsmith@yahoo.com~डी एस एम आई टी एच एट याहू डॉट कॉम -hgarza@gmail.com~एच जी ए आर ज़ेड ए एट जीमेल डॉट कॉम -qhill@yahoo.com~क्यू एच आई एल एल एट याहू डॉट कॉम -green-turner.org~ग्रीन हाइफ़न टी यू आर एन ई आर डॉट ऑर्ग -sharma-badami.com~एस एच ए आर एम ए हाइफ़न बी ए डी ए एम आई डॉट कॉम -osborne-gross.com~ओ एस बी ओ आर एन ई हाइफ़न जी आर ओ एस एस डॉट कॉम -lucero-stevenson.net~एल यू सी ई आर ओ हाइफ़न एस टी ई वी ई एन एस ओ एन डॉट नेट -https://google.com/~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश गूगल डॉट कॉम फॉरवर्ड स्लैश -https://github.com/~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश गिटहब डॉट कॉम फॉरवर्ड स्लैश -https://wikipedia.org/~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश विकिपीडिया डॉट ऑर्ग फॉरवर्ड स्लैश -https://amazon.com/~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश अमेज़ोन डॉट कॉम फॉरवर्ड स्लैश -www.google.com~डब्ल्यू डब्ल्यू डब्ल्यू डॉट गूगल डॉट कॉम -https://www.ndtv.com~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश डब्ल्यू डब्ल्यू डब्ल्यू डॉट एन डी टी वी डॉट कॉम -https://www.rbi.org.in/~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश डब्ल्यू डब्ल्यू डब्ल्यू डॉट आर बी आई डॉट ऑर्ग डॉट इन फॉरवर्ड स्लैश -https://www.amity.edu~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश डब्ल्यू डब्ल्यू डब्ल्यू डॉट एमिटी डॉट ई डी यू -https://example.com/blog/~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश ई एक्स ए एम पी एल ई डॉट कॉम फॉरवर्ड स्लैश ब्लॉग फॉरवर्ड स्लैश -https://example.com/about.html~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश ई एक्स ए एम पी एल ई डॉट कॉम फॉरवर्ड स्लैश अबाउट डॉट एच टी एम एल -https://example.com/search.php~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश ई एक्स ए एम पी एल ई डॉट कॉम फॉरवर्ड स्लैश सर्च डॉट पी एच पी -http://ati.edu~एच टी टी पी कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश ए टी आई डॉट ई डी यू -http://gcu.edu~एच टी टी पी कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश जी सी यू डॉट ई डी यू -http://pima.edu~एच टी टी पी कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश पी आई एम ए डॉट ई डी यू -bamu.nic.in/~बी ए एम यू डॉट एन आई सी डॉट इन फॉरवर्ड स्लैश -bieap.gov.in/~बी आई ई ए पी डॉट जी ओ वी डॉट इन फॉरवर्ड स्लैश -www.sharda.ac.in~डब्ल्यू डब्ल्यू डब्ल्यू डॉट शारदा डॉट ए सी डॉट इन -C:\Users\HP\Desktop\~सी कोलन बैकवर्ड स्लैश यूज़र्स बैकवर्ड स्लैश एच पी बैकवर्ड स्लैश डेस्कटॉप बैकवर्ड स्लैश -C:\Users\HP\Downloads\~सी कोलन बैकवर्ड स्लैश यूज़र्स बैकवर्ड स्लैश एच पी बैकवर्ड स्लैश डाउनलोड्स बैकवर्ड स्लैश -C:\Users\HP\Documents\Zoom~सी कोलन बैकवर्ड स्लैश यूज़र्स बैकवर्ड स्लैश एच पी बैकवर्ड स्लैश डॉक्युमेंट्स बैकवर्ड स्लैश ज़ेड ओ ओ एम -/home/desktop~फॉरवर्ड स्लैश होम फॉरवर्ड स्लैश डेस्कटॉप -/etc/apache~फॉरवर्ड स्लैश ई टी सी फॉरवर्ड स्लैश अपाची -/var/www~फॉरवर्ड स्लैश वार फॉरवर्ड स्लैश डब्ल्यू डब्ल्यू डब्ल्यू +gmail.com~gmail डॉट com +yahoo.com~yahoo डॉट com +hotmail.com~hotmail डॉट com +google.com~google डॉट com +kumaar.org~kumaar डॉट org +kumaar.info~kumaar डॉट info +kumar@gmail.com~kumar एट gmail डॉट com +robin@hotmail.com~robin एट hotmail डॉट com +kapil@live.com~kapil एट live डॉट com +sneha@live.com~sneha एट live डॉट com +mayank@google.com~mayank एट google डॉट com +charu@yahoo.com~charu एट yahoo डॉट com +john20@yahoo.com~john दो शून्य एट yahoo डॉट com +vivaan62@gmail.com~vivaan छह दो एट gmail डॉट com +viaan15@kumaar.com~viaan एक पाँच एट kumaar डॉट com +ltaa12@gmail.com~ltaa एक दो एट gmail डॉट com +kristen11@hotmail.com~kristen एक एक एट hotmail डॉट com +dsmith@yahoo.com~dsmith एट yahoo डॉट com +hgarza@gmail.com~hgarza एट gmail डॉट com +qhill@yahoo.com~qhill एट yahoo डॉट com +green-turner.org~green हाइफ़न turner डॉट org +sharma-badami.com~sharma हाइफ़न badami डॉट com +osborne-gross.com~osborne हाइफ़न gross डॉट com +lucero-stevenson.net~lucero हाइफ़न stevenson डॉट net +https://google.com/~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश google डॉट com फॉरवर्ड स्लैश +https://github.com/~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश github डॉट com फॉरवर्ड स्लैश +https://wikipedia.org/~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश wikipedia डॉट org फॉरवर्ड स्लैश +https://amazon.com/~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश amazon डॉट com फॉरवर्ड स्लैश +www.google.com~www डॉट google डॉट com +https://www.ndtv.com~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश www डॉट ndtv डॉट com +https://www.rbi.org.in/~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश www डॉट rbi डॉट org डॉट in फॉरवर्ड स्लैश +https://www.amity.edu~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश www डॉट amity डॉट edu +https://example.com/blog/~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश example डॉट com फॉरवर्ड स्लैश blog फॉरवर्ड स्लैश +https://example.com/about.html~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश example डॉट com फॉरवर्ड स्लैश about डॉट html +https://example.com/search.php~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश example डॉट com फॉरवर्ड स्लैश search डॉट php +http://ati.edu~http कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश ati डॉट edu +http://gcu.edu~http कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश gcu डॉट edu +http://pima.edu~http कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश pima डॉट edu +bamu.nic.in/~bamu डॉट nic डॉट in फॉरवर्ड स्लैश +bieap.gov.in/~bieap डॉट gov डॉट in फॉरवर्ड स्लैश +www.sharda.ac.in~www डॉट sharda डॉट ac डॉट in +C:\Users\HP\Desktop~C कोलन बैकवर्ड स्लैश Users बैकवर्ड स्लैश HP बैकवर्ड स्लैश Desktop +C:\Users\HP\Downloads~C कोलन बैकवर्ड स्लैश Users बैकवर्ड स्लैश HP बैकवर्ड स्लैश Downloads +C:\Users\HP\Documents\Zoom~C कोलन बैकवर्ड स्लैश Users बैकवर्ड स्लैश HP बैकवर्ड स्लैश Documents बैकवर्ड स्लैश Zoom +/home/desktop~फॉरवर्ड स्लैश home फॉरवर्ड स्लैश desktop +/etc/apache~फॉरवर्ड स्लैश etc फॉरवर्ड स्लैश apache +/var/www~फॉरवर्ड स्लैश var फॉरवर्ड स्लैश www 192.168.1.1~एक नौ दो डॉट एक छह आठ डॉट एक डॉट एक 10.0.0.1~एक शून्य डॉट शून्य डॉट शून्य डॉट एक 83.54.245.61~आठ तीन डॉट पाँच चार डॉट दो चार पाँच डॉट छह एक @@ -53,8 +53,13 @@ C:\Users\HP\Documents\Zoom~सी कोलन बैकवर्ड स्ल 255.255.255.0~दो पाँच पाँच डॉट दो पाँच पाँच डॉट दो पाँच पाँच डॉट शून्य आईपी पता है 192.168.1.1~आईपी पता है एक नौ दो डॉट एक छह आठ डॉट एक डॉट एक आईपी एड्रेस 10.0.0.1~आईपी एड्रेस एक शून्य डॉट शून्य डॉट शून्य डॉट एक -ip address 192.168.1.1~आई पी एड्रेस एक नौ दो डॉट एक छह आठ डॉट एक डॉट एक -ip address 10.0.0.1~आई पी एड्रेस एक शून्य डॉट शून्य डॉट शून्य डॉट एक -report.pdf~आर ई पी ओ आर टी डॉट पी डी एफ -photo.jpg~पी एच ओ टी ओ डॉट जे पी जी -data.csv~डेटा डॉट सी एस वी \ No newline at end of file +ip address 192.168.1.1~ip address एक नौ दो डॉट एक छह आठ डॉट एक डॉट एक +ip address 10.0.0.1~ip address एक शून्य डॉट शून्य डॉट शून्य डॉट एक +report.pdf~report डॉट pdf +photo.jpg~photo डॉट jpg +data.csv~data डॉट csv +robinson.org~robinson डॉट org +anand@gmail.com~anand एट gmail डॉट com +Al₂(SO₄)₃~ए एल दो ओपन ब्रेकेट एस ओ चार क्लोज़ ब्रेकेट तीन +C₂H₄~सी दो एच चार +home/desktop~home फॉरवर्ड स्लैश desktop \ No newline at end of file diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_fraction.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_fraction.txt index 4184ae9ee..6778978b7 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_fraction.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_fraction.txt @@ -20,4 +20,8 @@ १०००००००००००००००/८~एक पद्म बटा आठ 100000000000000000/412~एक शंख बटा चार सौ बारह २ २/७~दो और दो बटा सात -120 75/90~एक सौ बीस और पचहत्तर बटा नब्बे \ No newline at end of file +120 75/90~एक सौ बीस और पचहत्तर बटा नब्बे +१/२~आधा +१/३~तिहाई +1/4~चौथाई +3/4~तीन चौथाई \ No newline at end of file diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_measure.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_measure.txt index 6d0cef9b1..6afd66b7f 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_measure.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_measure.txt @@ -26,10 +26,6 @@ २५.४ °C~पच्चीस दशमलव चार डिग्री सेल्सियस 22 °F~बाईस डिग्री फारेनहाइट २२.५ °F~बाईस दशमलव पाँच डिग्री फारेनहाइट -7 K~सात केल्विन -७.२२ K~सात दशमलव दो दो केल्विन -5 L~पाँच लीटर -५.४ L~पाँच दशमलव चार लीटर 50 ml~पचास मिलीलीटर ५०.५ ml~पचास दशमलव पाँच मिलीलीटर 19 qt~उन्नीस क्वार्ट @@ -70,4 +66,4 @@ ५ yr~पाँच साल 1.5 yr~डेढ़ साल २.५ yr~ढाई साल -3.5 yr~साढ़े तीन साल +3.5 yr~साढ़े तीन साल \ No newline at end of file diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_money.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_money.txt index 0b199ff37..e5f157872 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_money.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_money.txt @@ -116,3 +116,29 @@ $९.९९~नौ डॉलर निन्यानबे सेंट ₦१०.२७~दस नाइरा सत्ताईस कोबो €200.90~दो सौ यूरो नब्बे सेंट €१२३४.७५~एक हज़ार दो सौ चौंतीस यूरो पचहत्तर सेंट +$1.12~एक डॉलर बारह सेंट +$1.123~एक दशमलव एक दो तीन डॉलर +$1.1234~एक दशमलव एक दो तीन चार डॉलर +₹2.2000~दो रुपए बीस पैसे +$1.2000~एक डॉलर बीस सेंट +₹1.500~एक रुपया पचास पैसे +₹5.00~पाँच रुपए +₹१~एक रुपया +₹२.१२३~दो दशमलव एक दो तीन रुपए +₹१.१२३४~एक दशमलव एक दो तीन चार रुपए +₹3,24,50,000~तीन करोड़ चौबीस लाख पचास हज़ार रुपए +₹32,450,000~तीन करोड़ चौबीस लाख पचास हज़ार रुपए +₹5,50,00,000~पाँच करोड़ पचास लाख रुपए +₹12,54,000~बारह लाख चौवन हज़ार रुपए +₹1,00,000~एक लाख रुपए +₹2,148~दो हज़ार एक सौ अड़तालीस रुपए +₹99,999~निन्यानबे हज़ार नौ सौ निन्यानबे रुपए +₹३,२४,५०,०००~तीन करोड़ चौबीस लाख पचास हज़ार रुपए +₹३२,४५०,०००~तीन करोड़ चौबीस लाख पचास हज़ार रुपए +₹२,१२,१५,०००~दो करोड़ बारह लाख पंद्रह हज़ार रुपए +₹५,५०,००,०००~पाँच करोड़ पचास लाख रुपए +₹१२,५४,०००~बारह लाख चौवन हज़ार रुपए +₹५,५६,३२०~पाँच लाख छप्पन हज़ार तीन सौ बीस रुपए +₹१,००,०००~एक लाख रुपए +₹२,१४८~दो हज़ार एक सौ अड़तालीस रुपए +₹९९,९९९~निन्यानबे हज़ार नौ सौ निन्यानबे रुपए \ No newline at end of file diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_roman.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_roman.txt new file mode 100644 index 000000000..340c754ed --- /dev/null +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_roman.txt @@ -0,0 +1,23 @@ +भास्कर-II~भास्कर दो +चंद्रयान-III~चंद्रयान तीन +अग्नि-IV~अग्नि चार +श्रेणी-II~श्रेणी दो +कक्षा XII~कक्षा बारह +अध्याय IV~अध्याय चार +भाग III~भाग तीन +खंड V~खंड पाँच +विश्व युद्ध II~विश्व युद्ध दो +विश्व युद्ध-II~विश्व युद्ध दो +प्रथम पंचवर्षीय योजना-I~प्रथम पंचवर्षीय योजना एक +राष्ट्रीय राजमार्ग-IV~राष्ट्रीय राजमार्ग चार +रोहिणी आर एस-I~रोहिणी आर एस एक +पीएसएलवी सी-IV~पीएसएलवी सी चार +ISRO मिशन-III~ISRO मिशन तीन +कक्षा XII की परीक्षा~कक्षा बारह की परीक्षा +XIIवीं कक्षा की परीक्षा~बारहवीं कक्षा की परीक्षा +भाग II का सारांश~भाग दो का सारांश +अध्याय IV के प्रश्न~अध्याय चार के प्रश्न +IVथी कक्षा के विद्यार्थी~चौथी कक्षा के विद्यार्थी +XC विद्यार्थी~नब्बे विद्यार्थी +LIII वा गणतंत्र दिन~तिरपन वा गणतंत्र दिन +भाग-XCIX~भाग निन्यानवे \ No newline at end of file diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_serial.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_serial.txt new file mode 100644 index 000000000..4a105554d --- /dev/null +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_serial.txt @@ -0,0 +1,31 @@ +कोविड-19~कोविड-उन्नीस +कोविड-१९~कोविड-उन्नीस +5जी~पाँच जी +५जी~पाँच जी +2^2~दो स्क्वेर्ड +२^२~दो स्क्वेर्ड +1-800-555~एक-आठ सौ-पाँच सौ पचपन +3जी~तीन जी +4जी~चार जी +कोरोना-2~कोरोना-दो +अग्नि-5~अग्नि-पाँच +ओमिक्रॉन-2~ओमिक्रॉन-दो +3^2~तीन स्क्वेर्ड +2^3~दो क्यूब +5^3~पाँच क्यूब +४^५~चार टु द पावर पाँच +99-1~निन्यानबे-एक +10-20-30~दस-बीस-तीस +1-800-999~एक-आठ सौ-नौ सौ निन्यानबे +पृथ्वी-4~पृथ्वी-चार +ब्रह्मोस-1~ब्रह्मोस-एक +Q1~क्यू एक +A10~ए दस +A12~ए बारह +B-60~बी-साठ +ABC-123~ए बी सी-एक सौ तेईस +FY2024~एफ वाई दो शून्य दो चार +H2O~एच दो ओ +CO2~सी ओ दो +ABCDE1234F~ए बी सी डी ई एक दो तीन चार एफ +F16~एफ सोलह \ No newline at end of file diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_word.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_word.txt index e9649919e..e7a284e9f 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_word.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_word.txt @@ -13,4 +13,8 @@ टाटा~टाटा ~ झ~झ -संगीत~संगीत \ No newline at end of file +संगीत~संगीत +This is a sentence.~This is a sentence. +google~google +mera email hai~mera email hai +user@~user@ \ No newline at end of file diff --git a/tests/nemo_text_processing/hi/test_roman.py b/tests/nemo_text_processing/hi/test_roman.py index e52d4897c..8d646d0f2 100644 --- a/tests/nemo_text_processing/hi/test_roman.py +++ b/tests/nemo_text_processing/hi/test_roman.py @@ -12,17 +12,29 @@ # See the License for the specific language governing permissions and # limitations under the License. + import pytest from parameterized import parameterized from nemo_text_processing.inverse_text_normalization.inverse_normalize import InverseNormalizer +from nemo_text_processing.text_normalization.normalize import Normalizer from ..utils import CACHE_DIR, parse_test_case_file class TestRoman: + normalizer = Normalizer( + input_case='cased', lang='hi', cache_dir=CACHE_DIR, overwrite_cache=False, post_process=False + ) inverse_normalizer = InverseNormalizer(lang='hi', cache_dir=CACHE_DIR, overwrite_cache=False) + @parameterized.expand(parse_test_case_file('hi/data_text_normalization/test_cases_roman.txt')) + @pytest.mark.run_only_on('CPU') + @pytest.mark.unit + def test_norm(self, test_input, expected): + pred = self.normalizer.normalize(test_input, verbose=False) + assert pred.strip() == expected.strip() + @parameterized.expand(parse_test_case_file('hi/data_inverse_text_normalization/test_cases_roman.txt')) @pytest.mark.run_only_on('CPU') @pytest.mark.unit diff --git a/tests/nemo_text_processing/hi/test_serial.py b/tests/nemo_text_processing/hi/test_serial.py index 8f5255df2..8302808c3 100644 --- a/tests/nemo_text_processing/hi/test_serial.py +++ b/tests/nemo_text_processing/hi/test_serial.py @@ -16,13 +16,24 @@ from parameterized import parameterized from nemo_text_processing.inverse_text_normalization.inverse_normalize import InverseNormalizer +from nemo_text_processing.text_normalization.normalize import Normalizer from ..utils import CACHE_DIR, parse_test_case_file class TestSerial: + normalizer = Normalizer( + input_case='cased', lang='hi', cache_dir=CACHE_DIR, overwrite_cache=False, post_process=True + ) inverse_normalizer = InverseNormalizer(lang='hi', cache_dir=CACHE_DIR, overwrite_cache=False) + @parameterized.expand(parse_test_case_file('hi/data_text_normalization/test_cases_serial.txt')) + @pytest.mark.run_only_on('CPU') + @pytest.mark.unit + def test_norm(self, test_input, expected): + pred = self.normalizer.normalize(test_input, verbose=False, punct_post_process=True) + assert pred == expected + @parameterized.expand(parse_test_case_file('hi/data_inverse_text_normalization/test_cases_serial.txt')) @pytest.mark.run_only_on('CPU') @pytest.mark.unit diff --git a/tests/nemo_text_processing/hi/test_sparrowhawk_normalization.sh b/tests/nemo_text_processing/hi/test_sparrowhawk_normalization.sh index e8057a126..74e1cf9c9 100644 --- a/tests/nemo_text_processing/hi/test_sparrowhawk_normalization.sh +++ b/tests/nemo_text_processing/hi/test_sparrowhawk_normalization.sh @@ -52,15 +52,15 @@ testTNDecimal() { # runtest $input #} -#testTNSerial() { -# input=$PROJECT_DIR/hi/data_text_normalization/test_cases_serial.txt -# runtest $input -#} +testTNSerial() { + input=$PROJECT_DIR/hi/data_text_normalization/test_cases_serial.txt + runtest $input +} -#testTNRoman() { -# input=$PROJECT_DIR/en/data_text_normalization/test_cases_roman.txt -# runtest $input -#} +testTNRoman() { + input=$PROJECT_DIR/hi/data_text_normalization/test_cases_roman.txt + runtest $input +} testTNElectronic() { input=$PROJECT_DIR/hi/data_text_normalization/test_cases_electronic.txt diff --git a/tests/nemo_text_processing/hi_en/test_address.py b/tests/nemo_text_processing/hi_en/test_address.py index 4f7dc3c51..82ad593bd 100644 --- a/tests/nemo_text_processing/hi_en/test_address.py +++ b/tests/nemo_text_processing/hi_en/test_address.py @@ -28,11 +28,11 @@ class TestAddress: @pytest.mark.unit def test_denorm_hi(self, test_input, expected): pred = self.inverse_normalizer.inverse_normalize(test_input, verbose=False) - assert pred.strip() == expected.strip() + assert pred == expected @parameterized.expand(parse_test_case_file('hi/data_inverse_text_normalization/test_cases_address.txt')) @pytest.mark.run_only_on('CPU') @pytest.mark.unit def test_denorm_hi_native(self, test_input, expected): pred = self.inverse_normalizer.inverse_normalize(test_input, verbose=False) - assert pred.strip() == expected.strip() + assert pred == expected diff --git a/tests/nemo_text_processing/ja/data_inverse_text_normalization/test_cases_decimal.txt b/tests/nemo_text_processing/ja/data_inverse_text_normalization/test_cases_decimal.txt index d5080acd0..4f91cf83a 100644 --- a/tests/nemo_text_processing/ja/data_inverse_text_normalization/test_cases_decimal.txt +++ b/tests/nemo_text_processing/ja/data_inverse_text_normalization/test_cases_decimal.txt @@ -1,32 +1,32 @@ -マイナス一点零六~-1.06 -マイナス七点零零六~-7.006 -マイナス三十九点五七四~-39.574 -マイナス三点八六~-3.86 -マイナス九十二点一五七四~-92.1574 -マイナス九点零三八~-9.038 -マイナス二点八七四一~-2.8741 -マイナス二百三十一点四六零九~-231.4609 -マイナス五十二点一八~-52.18 -マイナス五点三~-5.3 -マイナス五百七十九点三零零二~-579.3002 -マイナス八十六点四~-86.4 -マイナス八点四零九~-8.409 -マイナス八百二十一点七九五四~-821.7954 -マイナス八百五十二点七~-852.7 -マイナス六十一点零七~-61.07 -マイナス六点八一四~-6.814 -マイナス六百五十七点三零二四~-657.3024 -マイナス四十二点六零五~-42.605 -マイナス四百八十九点零五二一~-489.0521 -答えはマイナス一点零六~答えは-1.06 -計算の結果はマイナス七点零零六~計算の結果は-7.006 -マイナス二点八七四はかなり悪いスコア~-2.874はかなり悪いスコア -五点三は平均点~5.3は平均点 -テストの点数は八十六点四~テストの点数は86.4 -マイナス三十九点五七四は低すぎる~-39.574は低すぎる -答えはマイナス一点零六~答えは-1.06 -計算の結果はマイナス八十六点四~計算の結果は-86.4 -マイナス五十二点一八はかなり悪いスコア~-52.18はかなり悪いスコア -六点八一四は平均点~6.814は平均点 -テストの点数は九十二点一五七四~テストの点数は92.1574 -マイナス七点零零六は低すぎる~-7.006は低すぎる +マイナス一点ゼロ六~-1.06 +マイナス七点ゼロゼロ六~-7.006 +マイナス三十九点五七四~-39.574 +マイナス三点八六~-3.86 +マイナス九十二点一五七四~-92.1574 +マイナス九点ゼロ三八~-9.038 +マイナス二点八七四一~-2.8741 +マイナス二百三十一点四六ゼロ九~-231.4609 +マイナス五十二点一八~-52.18 +マイナス五点三~-5.3 +マイナス五百七十九点三ゼロゼロ二~-579.3002 +マイナス八十六点四~-86.4 +マイナス八点四ゼロ九~-8.409 +マイナス八百二十一点七九五四~-821.7954 +マイナス八百五十二点七~-852.7 +マイナス六十一点ゼロ七~-61.07 +マイナス六点八一四~-6.814 +マイナス六百五十七点三ゼロ二四~-657.3024 +マイナス四十二点六ゼロ五~-42.605 +マイナス四百八十九点ゼロ五二一~-489.0521 +答えはマイナス一点ゼロ六~答えは-1.06 +計算の結果はマイナス七点ゼロゼロ六~計算の結果は-7.006 +マイナス二点八七四はかなり悪いスコア~-2.874はかなり悪いスコア +五点三は平均点~5.3は平均点 +テストの点数は八十六点四~テストの点数は86.4 +マイナス三十九点五七四は低すぎる~-39.574は低すぎる +答えはマイナス一点ゼロ六~答えは-1.06 +計算の結果はマイナス八十六点四~計算の結果は-86.4 +マイナス五十二点一八はかなり悪いスコア~-52.18はかなり悪いスコア +六点八一四は平均点~6.814は平均点 +テストの点数は九十二点一五七四~テストの点数は92.1574 +マイナス七点ゼロゼロ六は低すぎる~-7.006は低すぎる diff --git a/tests/nemo_text_processing/ja/data_inverse_text_normalization/test_cases_fraction.txt b/tests/nemo_text_processing/ja/data_inverse_text_normalization/test_cases_fraction.txt index 32f80e812..d2abd62f6 100644 --- a/tests/nemo_text_processing/ja/data_inverse_text_normalization/test_cases_fraction.txt +++ b/tests/nemo_text_processing/ja/data_inverse_text_normalization/test_cases_fraction.txt @@ -1,34 +1,34 @@ -マイナス一と四分の三~-1 3/4 -一と四分の三~1 3/4 -マイナス一分の九~-9/1 -マイナス一分の六十~-60/1 -マイナス一分の百二十三~-123/1 -マイナス一荷四分の三~-1 3/4 -マイナス七百二十分の一~-1/720 -マイナス三十二分の三十一~-31/32 -マイナス三百九十七分の四~-4/397 -マイナス三百五十分の一~-1/350 -マイナス九十八分の四百七十一~-471/98 -マイナス二と五分の三~-2 3/5 -マイナス二十分の九~-9/20 -マイナス二十分の二十一~-21/20 -マイナス二十四分の一~-1/24 -マイナス二百二十分の一~-1/220 -マイナス二百五十二分の百四十七~-147/252 -マイナス二百五十六分の一~-1/256 -マイナス二荷五分の三~-2 3/5 -マイナス五分の七~-7/5 -マイナス五分の八~-8/5 -マイナス五分の十四~-14/5 -マイナス五分の百三十二~-132/5 -マイナス八分の五~-5/8 -答えはマイナス八分の五~答えは-5/8 -三分の一の人がその場を離れた~1/3の人がその場を離れた -約二分の一を削る~約1/2を削る -十分の三を削って吟醸をつくる~3/10を削って吟醸をつくる -一人三分の一ぐらい取る~1人1/3ぐらい取る -答えは九分の一~答えは1/9 -三分の二の人がその場を離れた~2/3の人がその場を離れた -約十分の一を削る~約1/10を削る -三分の一を削って吟醸をつくる~1/3を削って吟醸をつくる +マイナス一と四分の三~-1 3/4 +一と四分の三~1 3/4 +マイナス一分の九~-9/1 +マイナス一分の六十~-60/1 +マイナス一分の百二十三~-123/1 +マイナス一と四分の三~-1 3/4 +マイナス七百二十分の一~-1/720 +マイナス三十二分の三十一~-31/32 +マイナス三百九十七分の四~-4/397 +マイナス三百五十分の一~-1/350 +マイナス九十八分の四百七十一~-471/98 +マイナス二と五分の三~-2 3/5 +マイナス二十分の九~-9/20 +マイナス二十分の二十一~-21/20 +マイナス二十四分の一~-1/24 +マイナス二百二十分の一~-1/220 +マイナス二百五十二分の百四十七~-147/252 +マイナス二百五十六分の一~-1/256 +マイナス二と五分の三~-2 3/5 +マイナス五分の七~-7/5 +マイナス五分の八~-8/5 +マイナス五分の十四~-14/5 +マイナス五分の百三十二~-132/5 +マイナス八分の五~-5/8 +答えはマイナス八分の五~答えは-5/8 +三分の一の人がその場を離れた~1/3の人がその場を離れた +約二分の一を削る~約1/2を削る +十分の三を削って吟醸をつくる~3/10を削って吟醸をつくる +一人三分の一ぐらい取る~1人1/3ぐらい取る +答えは九分の一~答えは1/9 +三分の二の人がその場を離れた~2/3の人がその場を離れた +約十分の一を削る~約1/10を削る +三分の一を削って吟醸をつくる~1/3を削って吟醸をつくる 一人二分の一とぐらい取る~1人1/2とぐらい取る \ No newline at end of file diff --git a/tests/nemo_text_processing/ja/data_inverse_text_normalization/test_cases_time.txt b/tests/nemo_text_processing/ja/data_inverse_text_normalization/test_cases_time.txt index 6a5082124..14d0ab42e 100644 --- a/tests/nemo_text_processing/ja/data_inverse_text_normalization/test_cases_time.txt +++ b/tests/nemo_text_processing/ja/data_inverse_text_normalization/test_cases_time.txt @@ -1,40 +1,41 @@ -七時一分~7時1分 -七時四分~7時4分 -九時五十八分~9時58分 -九時十分前~9時10分前 -九時四十分~9時40分 -五時二十六分~5時26分 -六時五十五分~6時55分 -三時~3時 -三時~3時 -正午一分前~正午1分前 -正午十分過ぎ~正午10分過ぎ -九時三十分~9時30分 -七時五十分頃~7時50分頃 -一時~1時 -一時十分~1時10分 -三時~3時 -十七時~17時 -二十時~20時 -二十一時~21時 -二時~2時 -十二時三十分~12時30分 -零時~0時 -零時一分前~0時1分前 -二時~2時 -十二時~12時 -二十時~20時 -二十三時~23時 -二十四時~24時 -零時~0時 -四時~4時 -毎日五時に起きる~毎日5時に起きる -九時四十分の予約になります~9時40分の予約になります -現在の時間は十二時三十分~現在の時間は12時30分 -ちょうど零時になった~ちょうど0時になった -四時で店を閉める~4時で店を閉める -毎日六時に起きる~毎日6時に起きる -十時三十分の予約になります~10時30分の予約になります -現在の時間は十時三分~現在の時間は10時3分 -ちょうど一時になった~ちょうど1時になった -七時で店を閉める~7時で店を閉める +七時一分~7時1分 +七時四分~7時4分 +九時五十八分~9時58分 +九時十分前~9時10分前 +九時四十分~9時40分 +五時二十六分~5時26分 +六時五十五分~6時55分 +三時~3時 +三時~3時 +正午一分前~正午1分前 +正午十分過ぎ~正午10分過ぎ +九時三十分~9時30分 +七時五十分頃~7時50分頃 +一時~1時 +一時十分~1時10分 +三時~3時 +十七時~17時 +二十時~20時 +二十一時~21時 +二時~2時 +十二時三十分~12時30分 +ゼロ時~0時 +零時~0時 +ゼロ時一分前~0時1分前 +二時~2時 +十二時~12時 +二十時~20時 +二十三時~23時 +二十四時~24時 +ゼロ時~0時 +四時~4時 +毎日五時に起きる~毎日5時に起きる +九時四十分の予約になります~9時40分の予約になります +現在の時間は十二時三十分~現在の時間は12時30分 +ちょうどゼロ時になった~ちょうど0時になった +四時で店を閉める~4時で店を閉める +毎日六時に起きる~毎日6時に起きる +十時三十分の予約になります~10時30分の予約になります +現在の時間は十時三分~現在の時間は10時3分 +ちょうど一時になった~ちょうど1時になった +七時で店を閉める~7時で店を閉める diff --git a/tests/nemo_text_processing/ja/data_text_normalization/test_cases_address.txt b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_address.txt new file mode 100644 index 000000000..4e10ad910 --- /dev/null +++ b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_address.txt @@ -0,0 +1,9 @@ +東京都千代田区丸の内1-1-1~東京都千代田区丸の内一の一の一 +大阪府大阪市北区梅田3-1-1~大阪府大阪市北区梅田三の一の一 +神奈川県横浜市西区みなとみらい2-2-1~神奈川県横浜市西区みなとみらい二の二の一 +1丁目2番3号~一丁目二番三号 +3丁目5番1号~三丁目五番一号 +503号室~五〇三号室 +25階~二十五階 +〒100-0001~郵便番号一ゼロゼロのゼロゼロゼロ一 +住所は東京都港区六本木6-10-1です。~住所は東京都港区六本木六の十の一です。 diff --git a/tests/nemo_text_processing/ja/data_text_normalization/test_cases_cardinal.txt b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_cardinal.txt index 1d8a2801a..44a159017 100644 --- a/tests/nemo_text_processing/ja/data_text_normalization/test_cases_cardinal.txt +++ b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_cardinal.txt @@ -36,7 +36,7 @@ お年玉50000あげる~お年玉五万あげる 500000000000円分の株式を買った~五千億円分の株式を買った 今年の収益は100000000000になる~今年の収益は一千億になる -隣の会社の年収益は990000000000だそうだ~隣の会社の年収益は九千九百億だそうだ +隣の会社の収益は990000000000だそうだ~隣の会社の収益は九千九百億だそうだ 政府は100000000000の赤字で困っている~政府は一千億の赤字で困っている 兵士500人を派遣する~兵士五百人を派遣する お寺に10000寄付した~お寺に一万寄付した diff --git a/tests/nemo_text_processing/ja/data_text_normalization/test_cases_date.txt b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_date.txt index 6625ad4d0..b6eb392bd 100644 --- a/tests/nemo_text_processing/ja/data_text_normalization/test_cases_date.txt +++ b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_date.txt @@ -7,7 +7,6 @@ R.4~令和四年 1日から来年2月末まで~一日から来年二月末まで 1月〜12月~一月から十二月 1月の最終金曜日~一月の最終金曜日 -1月1日(月)〜3日(水)~一月一日月曜日から三日水曜日 1月22日~一月二十二日 70〜80年代~七十から八十年代 70年代~七十年代 diff --git a/tests/nemo_text_processing/ja/data_text_normalization/test_cases_decimal.txt b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_decimal.txt index 622b61d1a..a1e6f8c13 100644 --- a/tests/nemo_text_processing/ja/data_text_normalization/test_cases_decimal.txt +++ b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_decimal.txt @@ -12,4 +12,4 @@ 今年の冬の平均気温は-4.2度でした。~今年の冬の平均気温はマイナス四点二度でした。 昨日の株価は-3.6ポイント下落しました。~昨日の株価はマイナス三点六ポイント下落しました。 彼の体重は-2.4キログラム減りました。~彼の体重はマイナス二点四キログラム減りました。 -その製品の評価は-1.5ポイントでした。~その製品の評価はマイナス一点五ポイントでした。 \ No newline at end of file +その製品の評価は-1.5ポイントでした。~その製品の評価はマイナス一点五ポイントでした。 diff --git a/tests/nemo_text_processing/ja/data_text_normalization/test_cases_electronic.txt b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_electronic.txt new file mode 100644 index 000000000..086082b5f --- /dev/null +++ b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_electronic.txt @@ -0,0 +1,35 @@ +a@hotmail.de~a アット hotmail ドット de +a@hotmail.fr~a アット hotmail ドット fr +a@hotmail.it~a アット hotmail ドット it +a@aol.it~a アット aol ドット it +a@msn.it~a アット msn ドット it +abc@nvidia.app~abc アット nvidia ドット app +nvidia.co.jp~nvidia ドット co ドット jp +a.bc@gmail.com~a ドット bc アット gmail ドット com +cdf@abc.edu~cdf アット abc ドット edu +abc@gmail.abc~abc アット gmail ドット abc +abc@abc.com~abc アット abc ドット com +asdf123@abc.com~asdf 一二三 アット abc ドット com +ab3.sdd.3@gmail.com~ab 三 ドット sdd ドット 3 アット gmail ドット com +ab3-sdd-3@gmail.com~ab 三 ハイフン sdd ハイフン 3 アット gmail ドット com +www.nvidia.com~www ドット nvidia ドット com +nvidia.com~nvidia ドット com +google.co.jp~google ドット co ドット jp +www.google.co.jp~www ドット google ドット co ドット jp +nvidia.ai~nvidia ドット ai +http://www.nvidia.com~http コロン スラッシュ スラッシュ www ドット nvidia ドット com +https://www.nvidia.com~https コロン スラッシュ スラッシュ www ドット nvidia ドット com +https://developer.nvidia.com/drive-cuda/early-access~https コロン スラッシュ スラッシュ developer ドット nvidia ドット com スラッシュ drive ハイフン cuda スラッシュ early ハイフン access +abc@abc.com.~abc アット abc ドット com. +1234-5678-9012-3456~一二三四 五六七八 九〇一二 三四五六 +2345-2222-3333-4444~二三四五 二二二二 三三三三 四四四四 +9090-1234-5555-9876~九〇九〇 一二三四 五五五五 九八七六 +カード末尾3456~カード末尾三四五六 +カード下4桁7890~カード下四桁七八九〇 +カード番号1234-5678-9012-3456~カード番号一二三四 五六七八 九〇一二 三四五六 +写真.jpg~写真 ドット jpg +写真.JPG~写真 ドット JPG +写真.png~写真 ドット png +写真.PNG~写真 ドット PNG +資料.pdf~資料 ドット pdf +資料.PDF~資料 ドット PDF diff --git a/tests/nemo_text_processing/ja/data_text_normalization/test_cases_fraction.txt b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_fraction.txt index e2095fbfa..0d1ca7aa3 100644 --- a/tests/nemo_text_processing/ja/data_text_normalization/test_cases_fraction.txt +++ b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_fraction.txt @@ -1,18 +1,9 @@ 1/2~二分の一 -1/2~マイナス二分の一 -1 1/2~一荷二分の一 +1 1/2~一と二分の一 1と1/2~一と二分の一 -1荷1/2~一荷二分の一 --1荷1/2~マイナス一荷二分の一 -マイナス1荷1/2~マイナス一荷二分の一 -マイナス√1荷1/2~マイナスルート一荷二分の一 --√1荷1/2~マイナスルート一荷二分の一 --1荷√1/2~マイナス一荷二分のルート一 3分の1~三分の一 -3分の1~マイナス三分の一 -√3分の1~マイナスルート三分の一 -1荷√1/2~一荷二分のルート一 -1荷√1/3~一荷三分のルート一 1と1/3~一と三分の一 -1荷√1/4~一荷四分のルート一 1と√1/4~一と四分のルート一 diff --git a/tests/nemo_text_processing/ja/data_text_normalization/test_cases_measure.txt b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_measure.txt new file mode 100644 index 000000000..6eab137ce --- /dev/null +++ b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_measure.txt @@ -0,0 +1,34 @@ +5kg~五キロ +0kg~ゼロキロ +1 kg~一キロ +2kg~二キロ +100g~百グラム +500mg~五百ミリグラム +-3kg~マイナス三キロ +1.5kg~一点五キロ +-2.5kg~マイナス二点五キロ +0.05m~零点零五メートル +2.5cm~二点五センチ +3.14m2~三点一四平方メートル +3.14m²~三点一四平方メートル +1/2L~二分の一リットル +-1/2L~マイナス二分の一リットル +3分の1m~三分の一メートル +60km/h~時速六十キロ +60 km / h~時速六十キロ +60kg/h~六十キロ毎時 +50m/s~秒速五十メートル +50m/s~秒速五十メートル +100%~百パーセント +100%~百パーセント +30°C~三十度 +5μg~五マイクログラム +250mL~二百五十ミリリットル +440Hz~四百四十ヘルツ +220V~二百二十ボルト +10GB~十ギガバイト +5kg.~五キロ. +重さ 5kgです。~重さ五キロです。 +この箱は5kgです。~この箱は五キロです。 +温度は30°C。~温度は三十度。 +容量は1.5L、重さは500gです。~容量は一点五リットル、重さは五百グラムです。 diff --git a/tests/nemo_text_processing/ja/data_text_normalization/test_cases_money.txt b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_money.txt new file mode 100644 index 000000000..b930dda27 --- /dev/null +++ b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_money.txt @@ -0,0 +1,38 @@ +100円~百円 +1円~一円 +0円~ゼロ円 +10円~十円 +1,000円~千円 +12,345円~一万二千三百四十五円 +-500円~マイナス五百円 +マイナス500円~マイナス五百円 +¥100~百円 +¥2500~二千五百円 +JPY 3000~三千円 +jpy400~四百円 +¥3万~三万円 +1.5万円~一点五万円 +2億円~二億円 +3兆円~三兆円 +12円50銭~十二円五十銭 +¥12.50~十二円五十銭 +¥12.05~十二円五銭 +$5~五ドル +US$ 12~十二ドル +USD99~九十九ドル +usd 1.25~一ドル二十五セント +5ドル25セント~五ドル二十五セント +5米ドル25セント~五ドル二十五セント +$12.50~十二ドル五十セント +$0.99~ゼロドル九十九セント +-US$10.50~マイナス十ドル五十セント +€7~七ユーロ +EUR 8.40~八ユーロ四十セント +eur9~九ユーロ +£15~十五ポンド +GBP 20~二十ポンド +gbp30~三十ポンド +₩1000~千ウォン +KRW 5000~五千ウォン +CNY 10~十元 +100人民元~百人民元 diff --git a/tests/nemo_text_processing/ja/data_text_normalization/test_cases_range.txt b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_range.txt new file mode 100644 index 000000000..3778b9993 --- /dev/null +++ b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_range.txt @@ -0,0 +1,18 @@ +2-5~二から五 +10〜20~十から二十 +10〜20人~十から二十人 +3〜5歳~三から五歳 +10-15分~十から十五分 +1-3日~一から三日 +1000-2000円~千から二千円 +5000〜8000円~五千から八千円 +3kg-6kg~三キロから六キロ +15cm-25cm~十五センチから二十五センチ +10:00-11:00~十時から十一時 +9:00-10:30~九時から十時三十分 +10%〜15%~十パーセントから十五パーセント +2x3~二かける三 +70〜80年代~七十から八十年代 +3〜4月~三から四月 +1960s-1980s~千九百六十年代から千九百八十年代 +平成20〜30年代~平成二十から三十年代 diff --git a/tests/nemo_text_processing/ja/data_text_normalization/test_cases_roman.txt b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_roman.txt new file mode 100644 index 000000000..ad6617b6c --- /dev/null +++ b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_roman.txt @@ -0,0 +1,9 @@ +第III章~第三章 +第IV条~第四条 +第XII巻~第十二巻 +Chapter IV~Chapter 四 +第IX回~第九回 +Part III~Part 三 +第v条~第五条 +chapter ix~chapter 九 +Century XXI~Century 二十一 diff --git a/tests/nemo_text_processing/ja/data_text_normalization/test_cases_serial.txt b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_serial.txt new file mode 100644 index 000000000..a5e89a2e9 --- /dev/null +++ b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_serial.txt @@ -0,0 +1,14 @@ +B2A23C~ビー 二 エー 二三 シー +C24~シー 二四 +W2s~ダブリュー 二 エス +2x~二 エックス +covid-19~コビッド 十九 +133-ABC~一三三 ハイフン エービーシー +MIG-25/235212-asdg~エムアイジー ハイフン 二五 スラッシュ 二三五二一二 ハイフン エーエスディージー +JL123~ジェーエル 一二三 +NH456~エヌエイチ 四五六 +CX-5~シーエックス ハイフン 五 +型番ABC-1234~型番 エービーシー ハイフン 一二三四 +〒100-0001~郵便番号一ゼロゼロのゼロゼロゼロ一 +ISBN978-4-123456-78-9~アイエスビーエヌ 九七八 ハイフン 四 ハイフン 一二三四五六 ハイフン 七八 ハイフン 九 +3号車~三号車 diff --git a/tests/nemo_text_processing/ja/data_text_normalization/test_cases_telephone.txt b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_telephone.txt new file mode 100644 index 000000000..8f0680075 --- /dev/null +++ b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_telephone.txt @@ -0,0 +1,30 @@ +090-1234-5678~ゼロ九ゼロ、 一二三四、 五六七八 +080.1234.5678~ゼロ八ゼロ、 一二三四、 五六七八 +090-1234-5678~ゼロ九ゼロ、 一二三四、 五六七八 +03-1234-5678~ゼロ三、 一二三四、 五六七八 +(03) 1234-5678~ゼロ三、 一二三四、 五六七八 +045-123-4567~ゼロ四五、 一二三、 四五六七 +0120-123-456~ゼロ一二ゼロ、 一二三、 四五六 +050-1234-5678~ゼロ五ゼロ、 一二三四、 五六七八 ++81 90-1234-5678~プラス八一、 九ゼロ、 一二三四、 五六七八 ++81-90-1234-5678~プラス八一、 九ゼロ、 一二三四、 五六七八 +090-1234-5678.~ゼロ九ゼロ、 一二三四、 五六七八. +電話番号は090-1234-5678です。~電話番号はゼロ九ゼロ、 一二三四、 五六七八です。 +09012345678~ゼロ九ゼロ一二三四五六七八 +090 1234 5678~ゼロ九ゼロ、 一二三四、 五六七八 +(090) 1234-5678~ゼロ九ゼロ、 一二三四、 五六七八 +0570-123-456~ゼロ五七ゼロ、 一二三、 四五六 ++81 (3) 1234-5678~プラス八一、 三、 一二三四、 五六七八 ++1 (415) 555-0123~プラス一、 四一五、 五五五、 ゼロ一二三 +0800-123-4567~ゼロ八ゼロゼロ、 一二三、 四五六七 +0120-1234-567~ゼロ一二ゼロ、 一二三四、 五六七 +082-123-4567~ゼロ八二、 一二三、 四五六七 +0466-12-3456~ゼロ四六六、 一二、 三四五六 ++81 82-123-4567~プラス八一、 八二、 一二三、 四五六七 ++81 (45) 123-4567~プラス八一、 四五、 一二三、 四五六七 ++44 20-7946-0958~プラス四四、 二ゼロ、 七九四六、 ゼロ九五八 ++84 (28) 3822-9999~プラス八四、 二八、 三八二二、 九九九九 +111-222-3333~一一一、 二二二、 三三三三 +03-1234-5678 内線123~ゼロ三、 一二三四、 五六七八、 内線 一二三 +090-1234-5678に電話してください。~ゼロ九ゼロ、 一二三四、 五六七八に電話してください。 +0120-123-456に電話~ゼロ一二ゼロ、 一二三、 四五六に電話 diff --git a/tests/nemo_text_processing/ja/data_text_normalization/test_cases_time.txt b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_time.txt index 42e7600a7..ca805d214 100644 --- a/tests/nemo_text_processing/ja/data_text_normalization/test_cases_time.txt +++ b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_time.txt @@ -199,4 +199,4 @@ 翌日27時~翌日二十七時 翌日25時~翌日二十五時 翌日26時~翌日二十六時 -0時~零時 \ No newline at end of file +0時~零時 diff --git a/tests/nemo_text_processing/ja/data_text_normalization/test_cases_whitelist.txt b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_whitelist.txt new file mode 100644 index 000000000..1f637ed7e --- /dev/null +++ b/tests/nemo_text_processing/ja/data_text_normalization/test_cases_whitelist.txt @@ -0,0 +1,14 @@ +Dr.田中~ドクター田中 +Mr.山田~ミスター山田 +Mrs.佐藤~ミセス佐藤 +Ms.鈴木~ミス鈴木 +Prof. Smith~プロフェッサー Smith +vs.~バーサス +No.3~ナンバー三 +etc.~エトセトラ +Ph.D.~p h d +St.~ストリート +Mt.~マウント +Ave.~アベニュー +jr.~ジュニア +Sr.~シニア diff --git a/tests/nemo_text_processing/ja/test_address.py b/tests/nemo_text_processing/ja/test_address.py new file mode 100644 index 000000000..535791054 --- /dev/null +++ b/tests/nemo_text_processing/ja/test_address.py @@ -0,0 +1,31 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pytest +from parameterized import parameterized + +from nemo_text_processing.text_normalization.normalize import Normalizer + +from ..utils import CACHE_DIR, parse_test_case_file + + +class TestAddress: + normalizer_ja = Normalizer(lang='ja', cache_dir=CACHE_DIR, overwrite_cache=False, input_case='cased') + + @parameterized.expand(parse_test_case_file('ja/data_text_normalization/test_cases_address.txt')) + @pytest.mark.run_only_on('CPU') + @pytest.mark.unit + def test_norm_address(self, test_input, expected): + preds = self.normalizer_ja.normalize(test_input) + assert expected == preds diff --git a/tests/nemo_text_processing/ja/test_date.py b/tests/nemo_text_processing/ja/test_date.py index cd7127698..d6125b842 100644 --- a/tests/nemo_text_processing/ja/test_date.py +++ b/tests/nemo_text_processing/ja/test_date.py @@ -37,6 +37,6 @@ def test_denorm(self, test_input, expected): @parameterized.expand(parse_test_case_file('ja/data_text_normalization/test_cases_date.txt')) @pytest.mark.run_only_on('CPU') @pytest.mark.unit - def test_denorm(self, test_input, expected): + def test_norm(self, test_input, expected): pred = self.normalizer.normalize(test_input, verbose=False) assert pred == expected diff --git a/tests/nemo_text_processing/ja/test_electronic.py b/tests/nemo_text_processing/ja/test_electronic.py new file mode 100644 index 000000000..a2665c1ae --- /dev/null +++ b/tests/nemo_text_processing/ja/test_electronic.py @@ -0,0 +1,31 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pytest +from parameterized import parameterized + +from nemo_text_processing.text_normalization.normalize import Normalizer + +from ..utils import CACHE_DIR, parse_test_case_file + + +class TestElectronic: + normalizer_ja = Normalizer(lang='ja', cache_dir=CACHE_DIR, overwrite_cache=False, input_case='cased') + + @parameterized.expand(parse_test_case_file('ja/data_text_normalization/test_cases_electronic.txt')) + @pytest.mark.run_only_on('CPU') + @pytest.mark.unit + def test_norm_electronic(self, test_input, expected): + preds = self.normalizer_ja.normalize(test_input) + assert expected == preds diff --git a/tests/nemo_text_processing/ja/test_measure.py b/tests/nemo_text_processing/ja/test_measure.py new file mode 100644 index 000000000..dc2489427 --- /dev/null +++ b/tests/nemo_text_processing/ja/test_measure.py @@ -0,0 +1,32 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + + +import pytest +from parameterized import parameterized + +from nemo_text_processing.text_normalization.normalize import Normalizer + +from ..utils import CACHE_DIR, parse_test_case_file + + +class TestMeasure: + normalizer_ja = Normalizer(lang='ja', cache_dir=CACHE_DIR, overwrite_cache=False, input_case='cased') + + @parameterized.expand(parse_test_case_file('ja/data_text_normalization/test_cases_measure.txt')) + @pytest.mark.run_only_on('CPU') + @pytest.mark.unit + def test_norm_measure(self, test_input, expected): + preds = self.normalizer_ja.normalize(test_input) + assert expected == preds diff --git a/tests/nemo_text_processing/ja/test_money.py b/tests/nemo_text_processing/ja/test_money.py new file mode 100644 index 000000000..df4c04d6a --- /dev/null +++ b/tests/nemo_text_processing/ja/test_money.py @@ -0,0 +1,32 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + + +import pytest +from parameterized import parameterized + +from nemo_text_processing.text_normalization.normalize import Normalizer + +from ..utils import CACHE_DIR, parse_test_case_file + + +class TestMoney: + normalizer_ja = Normalizer(lang='ja', cache_dir=CACHE_DIR, overwrite_cache=False, input_case='cased') + + @parameterized.expand(parse_test_case_file('ja/data_text_normalization/test_cases_money.txt')) + @pytest.mark.run_only_on('CPU') + @pytest.mark.unit + def test_norm_money(self, test_input, expected): + preds = self.normalizer_ja.normalize(test_input) + assert expected == preds diff --git a/tests/nemo_text_processing/ja/test_range.py b/tests/nemo_text_processing/ja/test_range.py new file mode 100644 index 000000000..5b24746d5 --- /dev/null +++ b/tests/nemo_text_processing/ja/test_range.py @@ -0,0 +1,31 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pytest +from parameterized import parameterized + +from nemo_text_processing.text_normalization.normalize import Normalizer + +from ..utils import CACHE_DIR, parse_test_case_file + + +class TestRange: + normalizer_ja = Normalizer(lang='ja', cache_dir=CACHE_DIR, overwrite_cache=False, input_case='cased') + + @parameterized.expand(parse_test_case_file('ja/data_text_normalization/test_cases_range.txt')) + @pytest.mark.run_only_on('CPU') + @pytest.mark.unit + def test_norm_range(self, test_input, expected): + preds = self.normalizer_ja.normalize(test_input) + assert expected == preds diff --git a/tests/nemo_text_processing/ja/test_roman.py b/tests/nemo_text_processing/ja/test_roman.py new file mode 100644 index 000000000..289625c38 --- /dev/null +++ b/tests/nemo_text_processing/ja/test_roman.py @@ -0,0 +1,31 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pytest +from parameterized import parameterized + +from nemo_text_processing.text_normalization.normalize import Normalizer + +from ..utils import CACHE_DIR, parse_test_case_file + + +class TestRoman: + normalizer_ja = Normalizer(lang='ja', cache_dir=CACHE_DIR, overwrite_cache=False, input_case='cased') + + @parameterized.expand(parse_test_case_file('ja/data_text_normalization/test_cases_roman.txt')) + @pytest.mark.run_only_on('CPU') + @pytest.mark.unit + def test_norm_roman(self, test_input, expected): + preds = self.normalizer_ja.normalize(test_input) + assert expected == preds diff --git a/tests/nemo_text_processing/ja/test_serial.py b/tests/nemo_text_processing/ja/test_serial.py new file mode 100644 index 000000000..f991251dd --- /dev/null +++ b/tests/nemo_text_processing/ja/test_serial.py @@ -0,0 +1,31 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pytest +from parameterized import parameterized + +from nemo_text_processing.text_normalization.normalize import Normalizer + +from ..utils import CACHE_DIR, parse_test_case_file + + +class TestSerial: + normalizer_ja = Normalizer(lang='ja', cache_dir=CACHE_DIR, overwrite_cache=False, input_case='cased') + + @parameterized.expand(parse_test_case_file('ja/data_text_normalization/test_cases_serial.txt')) + @pytest.mark.run_only_on('CPU') + @pytest.mark.unit + def test_norm_serial(self, test_input, expected): + preds = self.normalizer_ja.normalize(test_input) + assert expected == preds diff --git a/tests/nemo_text_processing/ja/test_sparrowhawk_normalization.sh b/tests/nemo_text_processing/ja/test_sparrowhawk_normalization.sh index 42db11fd6..e8f676337 100644 --- a/tests/nemo_text_processing/ja/test_sparrowhawk_normalization.sh +++ b/tests/nemo_text_processing/ja/test_sparrowhawk_normalization.sh @@ -46,6 +46,42 @@ testTNDateText() { input=$PROJECT_DIR/ja/data_text_normalization/test_cases_date.txt runtest $input } +testTNMoneyText() { + input=$PROJECT_DIR/ja/data_text_normalization/test_cases_money.txt + runtest $input +} +testTNMeasureText() { + input=$PROJECT_DIR/ja/data_text_normalization/test_cases_measure.txt + runtest $input +} +testTNTelephoneText() { + input=$PROJECT_DIR/ja/data_text_normalization/test_cases_telephone.txt + runtest $input +} +testTNElectronicText() { + input=$PROJECT_DIR/ja/data_text_normalization/test_cases_electronic.txt + runtest $input +} +testTNSerialText() { + input=$PROJECT_DIR/ja/data_text_normalization/test_cases_serial.txt + runtest $input +} +testTNRangeText() { + input=$PROJECT_DIR/ja/data_text_normalization/test_cases_range.txt + runtest $input +} +testTNAddressText() { + input=$PROJECT_DIR/ja/data_text_normalization/test_cases_address.txt + runtest $input +} +testTNRomanText() { + input=$PROJECT_DIR/ja/data_text_normalization/test_cases_roman.txt + runtest $input +} +testTNWhitelistText() { + input=$PROJECT_DIR/ja/data_text_normalization/test_cases_whitelist.txt + runtest $input +} # Load shUnit2 diff --git a/tests/nemo_text_processing/ja/test_telephone.py b/tests/nemo_text_processing/ja/test_telephone.py new file mode 100644 index 000000000..4fcd8e116 --- /dev/null +++ b/tests/nemo_text_processing/ja/test_telephone.py @@ -0,0 +1,32 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + + +import pytest +from parameterized import parameterized + +from nemo_text_processing.text_normalization.normalize import Normalizer + +from ..utils import CACHE_DIR, parse_test_case_file + + +class TestTelephone: + normalizer_ja = Normalizer(lang='ja', cache_dir=CACHE_DIR, overwrite_cache=False, input_case='cased') + + @parameterized.expand(parse_test_case_file('ja/data_text_normalization/test_cases_telephone.txt')) + @pytest.mark.run_only_on('CPU') + @pytest.mark.unit + def test_norm_telephone(self, test_input, expected): + preds = self.normalizer_ja.normalize(test_input) + assert expected == preds diff --git a/tests/nemo_text_processing/ja/test_whitelist.py b/tests/nemo_text_processing/ja/test_whitelist.py new file mode 100644 index 000000000..ece7bbe63 --- /dev/null +++ b/tests/nemo_text_processing/ja/test_whitelist.py @@ -0,0 +1,31 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pytest +from parameterized import parameterized + +from nemo_text_processing.text_normalization.normalize import Normalizer + +from ..utils import CACHE_DIR, parse_test_case_file + + +class TestWhitelist: + normalizer_ja = Normalizer(lang='ja', cache_dir=CACHE_DIR, overwrite_cache=False, input_case='cased') + + @parameterized.expand(parse_test_case_file('ja/data_text_normalization/test_cases_whitelist.txt')) + @pytest.mark.run_only_on('CPU') + @pytest.mark.unit + def test_norm_whitelist(self, test_input, expected): + preds = self.normalizer_ja.normalize(test_input) + assert expected == preds diff --git a/tests/nemo_text_processing/ko/data_text_normalization/test_cases_cardinal.txt b/tests/nemo_text_processing/ko/data_text_normalization/test_cases_cardinal.txt index 40187f74e..dbde76d4e 100644 --- a/tests/nemo_text_processing/ko/data_text_normalization/test_cases_cardinal.txt +++ b/tests/nemo_text_processing/ko/data_text_normalization/test_cases_cardinal.txt @@ -45,4 +45,24 @@ -2~마이너스 이 -93~마이너스 구십삼 -90325~마이너스 구만삼백이십오 --3234567~마이너스 삼백이십삼만사천오백육십칠 \ No newline at end of file +-3234567~마이너스 삼백이십삼만사천오백육십칠 +휴대폰 번호는 0987654321입니다~휴대폰 번호는 영구팔칠육오사삼이일 입니다 +전화 번호는 01090817263입니다~전화 번호는 영일영구영팔일칠이육삼 입니다 +계좌 번호는 70501938462011입니다~계좌 번호는 칠영오영일구삼팔사육이영일일 입니다 +예약 번호는 907입니다~예약 번호는 구영칠 입니다 +예약 번호는 830-291입니다~예약 번호는 팔삼영이구일 입니다 +주문 번호는 4829.7301입니다~주문 번호는 사팔이구칠삼영일 입니다 +인증 번호는 000739입니다~인증 번호는 영영영칠삼구 입니다 +인증 번호는 204060입니다~인증 번호는 이영사영육영 입니다 +배송 번호는 591837462입니다~배송 번호는 오구일팔삼칠사육이 입니다 +회원 번호는 3000456789입니다~회원 번호는 삼영영영사오육칠팔구 입니다 +연락처는 84502719입니다~연락처는 팔사오영이칠일구 입니다 +연락처가 77008899입니다~연락처가 칠칠영영팔팔구구 입니다 +연락처를 604 812 930으로 저장했습니다~연락처를 육영사팔일이구삼영 으로 저장했습니다 +번호는 12입니다~번호는 십이 입니다 +배송 번호는 591837462입니다~배송 번호는 오구일팔삼칠사육이 입니다 +회원 번호는 3000456789입니다~회원 번호는 삼영영영사오육칠팔구 입니다 +접수 번호는 812709입니다~접수 번호는 팔일이칠영구 입니다 +신청 번호는 40682015입니다~신청 번호는 사영육팔이영일오 입니다 +확인 번호는 970045입니다~확인 번호는 구칠영영사오 입니다 +관리 번호는 25081973입니다~관리 번호는 이오영팔일구칠삼 입니다 \ No newline at end of file diff --git a/tests/nemo_text_processing/ko/data_text_normalization/test_cases_telephone.txt b/tests/nemo_text_processing/ko/data_text_normalization/test_cases_telephone.txt index b6e573aec..a871e4f71 100644 --- a/tests/nemo_text_processing/ko/data_text_normalization/test_cases_telephone.txt +++ b/tests/nemo_text_processing/ko/data_text_normalization/test_cases_telephone.txt @@ -30,4 +30,10 @@ +82-123.456-7890~국가번호 팔이 일이삼 사오육 칠팔구영 111-222-3333~일일일 이이이 삼삼삼삼 909-808-7070~구영구 팔영팔 칠영칠영 -(555)555-5555~오오오 오오오 오오오오 \ No newline at end of file +(555)555-5555~오오오 오오오 오오오오 +02-123-4567~영이 일이삼 사오육칠 +02-1234-5678~영이 일이삼사 오육칠팔 +043-123-4567~영사삼 일이삼 사오육칠 +043-1234-5678~영사삼 일이삼사 오육칠팔 +(02) 1234-5678~영이 일이삼사 오육칠팔 +010 1234 5678~영일영 일이삼사 오육칠팔 \ No newline at end of file diff --git a/tools/text_processing_deployment/pynini_export.py b/tools/text_processing_deployment/pynini_export.py index 03705f2b6..73a4fc138 100644 --- a/tools/text_processing_deployment/pynini_export.py +++ b/tools/text_processing_deployment/pynini_export.py @@ -278,6 +278,7 @@ def parse_args(): from nemo_text_processing.text_normalization.ar.taggers.tokenize_and_classify import ( ClassifyFst as TNClassifyFst, ) + from nemo_text_processing.text_normalization.ar.verbalizers.verbalize import VerbalizeFst as TNVerbalizeFst elif args.language == 'it': from nemo_text_processing.text_normalization.it.taggers.tokenize_and_classify import ( ClassifyFst as TNClassifyFst,