diff --git a/Jenkinsfile b/Jenkinsfile index f5d114f84..daa32bda4 100644 --- a/Jenkinsfile +++ b/Jenkinsfile @@ -27,9 +27,9 @@ pipeline { HE_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/09-24-25-0' HY_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/03-12-24-0' MR_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/03-12-24-1' + HI_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/09-21-26-0' JA_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/07-17-26-0' KO_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/09-15-26-0' - HI_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/09-15-26-0' DEFAULT_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/06-08-23-0' } stages { diff --git a/nemo_text_processing/inverse_text_normalization/hi/data/electronic/__init__.py b/nemo_text_processing/inverse_text_normalization/hi/data/electronic/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/nemo_text_processing/inverse_text_normalization/hi/data/electronic/common_words.tsv b/nemo_text_processing/inverse_text_normalization/hi/data/electronic/common_words.tsv new file mode 100644 index 000000000..d8dcb1c95 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/data/electronic/common_words.tsv @@ -0,0 +1,83 @@ +Users यूज़र्स +User यूज़र +Desktop डेस्कटॉप +Downloads डाउनलोड्स +Documents डॉक्युमेंट्स +Music म्यूज़िक +Pictures पिक्चर्स +Videos वीडियोज़ +Audio ऑडियो +file फ़ाइल +files फाइल्स +chapter चैप्टर +python पाइथन +work वर्क +about अबाउट +index इंडेक्स +tags टैग्स +app ऐप +wiki विकी +docs डॉक्स +master मास्टर +config कॉन्फ़िग +email ई मेल +apache अपाची +kernel कर्नल +bin बिन +var वार +home होम +backups बैकअप्स +temp टेम्प +data डेटा +survey सर्वे +ward वार्ड +templates टेम्पलेट्स +impress इंप्रेस +office ऑफिस +libreoffice लिब्रे ऑफिस +and एंड +LICENSE लाइसेंस +venv वेन्व +blog ब्लॉग +blogs ब्लॉग्स +login लॉगिन +register रजिस्टर +search सर्च +category केटेगरी +categories केटेगरीज़ +post पोस्ट +posts पोस्ट्स +faq एफ ए क्यू +terms टर्म्स +privacy प्राइवेसी +main मेन +explore एक्सप्लोर +photos फ़ोटोज़ +images इमेजेज़ +web वेब +online ऑनलाइन +courses कोर्सेज़ +learn लर्न +learning लर्निंग +university यूनिवर्सिटी +academy अकेडमी +domain डोमेन +domains डोमेन्स +analysis अनैलिसिस +maps मैप्स +services सर्विसेज़ +sites साइट्स +activate एक्टिवेट +available अवेलेबल +enabled इनेबल्ड +secure सेक्योर +school स्कूल +homepage होमपेज +content कंटेन्ट +default डिफ़ॉल्ट +list लिस्ट +tag टैग +laptop लैपटॉप +phone फोन +play प्ले +world वर्ल्ड \ No newline at end of file diff --git a/nemo_text_processing/inverse_text_normalization/hi/data/electronic/domain.tsv b/nemo_text_processing/inverse_text_normalization/hi/data/electronic/domain.tsv new file mode 100644 index 000000000..e20dd96e2 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/data/electronic/domain.tsv @@ -0,0 +1,16 @@ +com कॉम +org ऑर्ग +net नेट +edu ई डी यू +gov जी ओ वी +biz बिज़ +info इन्फो +in इन +io आई ओ +ai ए आई +uk यू के +us यू एस +ac ए सी +res आर ई एस +nic एन आई सी +ernet ई आर नेट \ No newline at end of file diff --git a/nemo_text_processing/inverse_text_normalization/hi/data/electronic/letters.tsv b/nemo_text_processing/inverse_text_normalization/hi/data/electronic/letters.tsv new file mode 100644 index 000000000..b488dd490 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/data/electronic/letters.tsv @@ -0,0 +1,28 @@ +a ए +b बी +c सी +d डी +e ई +f एफ +f एफ़ +g जी +h एच +i आई +i आइ +j जे +k के +l एल +m एम +n एन +o ओ +p पी +q क्यू +r आर +s एस +t टी +u यू +v वी +w डब्ल्यू +x एक्स +y वाई +z ज़ेड \ No newline at end of file diff --git a/nemo_text_processing/inverse_text_normalization/hi/data/electronic/server_name.tsv b/nemo_text_processing/inverse_text_normalization/hi/data/electronic/server_name.tsv new file mode 100644 index 000000000..39f164ce6 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/data/electronic/server_name.tsv @@ -0,0 +1,27 @@ +gmail जीमेल +yahoo याहू +hotmail हॉटमेल +outlook आउटलुक +google गूगल +blogger ब्लॉगर +live लाइव +microsoft माइक्रोसॉफ्ट +facebook फ़ेसबुक +twitter ट्विटर +instagram इंस्टाग्राम +linkedin लिंक्डइन +youtube यूट्यूब +amazon अमेज़ोन +wikipedia विकिपीडिया +github गिटहब +reddit रेडिट +netflix नेटफ्लिक्स +spotify स्पॉटिफाई +apple एप्पल +samsung सैमसंग +nvidia एनविडिया +intel इंटेल +adobe अडोब +wordpress वर्डप्रेस +mozilla मॉज़िला +ebay ई बे \ No newline at end of file diff --git a/nemo_text_processing/inverse_text_normalization/hi/data/electronic/symbols.tsv b/nemo_text_processing/inverse_text_normalization/hi/data/electronic/symbols.tsv new file mode 100644 index 000000000..c2d15e04c --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/data/electronic/symbols.tsv @@ -0,0 +1,25 @@ +backslash बैकवर्ड स्लैश \\ +forwardslash फॉरवर्ड स्लैश / +dot डॉट . +dot DOT . +point प्वाइंट . +hyphen हाइफ़न - +hyphen हाइफन - +underscore अंडर स्कोर _ +at एट +x एक्स x +space स्पेस +openbracket ओपन ब्रेकेट ( +closebracket क्लोज़ ब्रेकेट ) +dollar डॉलर $ +and एंड and +www डब्ल्यू डब्ल्यू डब्ल्यू www +v वी v +tilde टिल्ड +tilde ~ +hashtag हैशटैग # +colon कोलन : +https एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश https:// +http एच टी टी पी कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश http:// +litslash / / +lithyphen - - \ No newline at end of file diff --git a/nemo_text_processing/inverse_text_normalization/hi/data/measure/equal_symbols.tsv b/nemo_text_processing/inverse_text_normalization/hi/data/measure/equal_symbols.tsv new file mode 100644 index 000000000..367cee658 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/data/measure/equal_symbols.tsv @@ -0,0 +1,7 @@ +इज़ इक्वल टू = +बराबर होता है = +इक्वल टू = +बराबर है = +इक्वल्स = +बराबर = +इक्वल = diff --git a/nemo_text_processing/inverse_text_normalization/hi/data/measure/math_symbols.tsv b/nemo_text_processing/inverse_text_normalization/hi/data/measure/math_symbols.tsv new file mode 100644 index 000000000..83f474f78 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/data/measure/math_symbols.tsv @@ -0,0 +1,11 @@ +प्लस + +जमा + +जोड़ + +माइनस - +घटा - +डिवाइडेड / +भाग / +मल्टीप्लाइड × +गुणा × +गुना × +इनटू × diff --git a/nemo_text_processing/inverse_text_normalization/hi/data/roman/__init__.py b/nemo_text_processing/inverse_text_normalization/hi/data/roman/__init__.py new file mode 100644 index 000000000..4fc25d0d3 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/data/roman/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/inverse_text_normalization/hi/data/roman/key_words.tsv b/nemo_text_processing/inverse_text_normalization/hi/data/roman/key_words.tsv new file mode 100644 index 000000000..208c2375c --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/data/roman/key_words.tsv @@ -0,0 +1,4 @@ +अध्याय +खंड +खण्ड +कक्षा diff --git a/nemo_text_processing/inverse_text_normalization/hi/data/roman/roman_numerals.tsv b/nemo_text_processing/inverse_text_normalization/hi/data/roman/roman_numerals.tsv new file mode 100644 index 000000000..f443fd3a9 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/data/roman/roman_numerals.tsv @@ -0,0 +1,13 @@ +I 1 +V 5 +X 10 +L 50 +C 100 +D 500 +M 1000 +IV 4 +IX 9 +XL 40 +XC 90 +CD 400 +CM 900 diff --git a/nemo_text_processing/inverse_text_normalization/hi/data/serial/__init__.py b/nemo_text_processing/inverse_text_normalization/hi/data/serial/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/nemo_text_processing/inverse_text_normalization/hi/data/serial/power.tsv b/nemo_text_processing/inverse_text_normalization/hi/data/serial/power.tsv new file mode 100644 index 000000000..c9bd9f96b --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/data/serial/power.tsv @@ -0,0 +1 @@ +टु द पावर ^ diff --git a/nemo_text_processing/inverse_text_normalization/hi/data/serial/power_special.tsv b/nemo_text_processing/inverse_text_normalization/hi/data/serial/power_special.tsv new file mode 100644 index 000000000..d612417a6 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/data/serial/power_special.tsv @@ -0,0 +1,2 @@ +स्क्वेर्ड ^2 +क्यूब ^3 diff --git a/nemo_text_processing/inverse_text_normalization/hi/data/telephone/extension.tsv b/nemo_text_processing/inverse_text_normalization/hi/data/telephone/extension.tsv new file mode 100644 index 000000000..4db2e5ff2 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/data/telephone/extension.tsv @@ -0,0 +1,4 @@ +एक्सटेंशन ext. +एक्स्टेंशन ext. +एक्सटेंशन नंबर ext. +एक्स्टेंशन नंबर ext. diff --git a/nemo_text_processing/inverse_text_normalization/hi/data/time/time_zone.tsv b/nemo_text_processing/inverse_text_normalization/hi/data/time/time_zone.tsv new file mode 100644 index 000000000..993bfdc94 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/data/time/time_zone.tsv @@ -0,0 +1,7 @@ +IST आई एस टी +IST भारतीय मानक समय +IST भारतीय समयानुसार +GMT जी एम टी +UTC यू टी सी +PST पी एस टी +EST ई एस टी diff --git a/nemo_text_processing/inverse_text_normalization/hi/graph_utils.py b/nemo_text_processing/inverse_text_normalization/hi/graph_utils.py index b002efa52..70364611c 100644 --- a/nemo_text_processing/inverse_text_normalization/hi/graph_utils.py +++ b/nemo_text_processing/inverse_text_normalization/hi/graph_utils.py @@ -34,6 +34,21 @@ NEMO_HI_DIGIT = pynini.union("०", "१", "२", "३", "४", "५", "६", "७", "८", "९").optimize() DEVANAGARI_DIGIT = ["०", "१", "२", "३", "४", "५", "६", "७", "८", "९"] +DIGIT_GLYPH_TO_ASCII = pynini.union( + *[pynini.cross(glyph, str(value)) for value, glyph in enumerate(DEVANAGARI_DIGIT)] +).optimize() +DIGIT_WORD_TO_DEVANAGARI = ( + pynini.string_file(get_abs_path("data/numbers/digit.tsv")).invert() + | pynini.string_file(get_abs_path("data/numbers/zero.tsv")).invert() +).optimize() + +# Devanagari characters (consonants, vowels, matras, signs) excluding the digits +# (0x0966-0x096F). Shared so classes like serial and electronic can reuse it. +DEVANAGARI_LETTER = pynini.union( + *[chr(c) for c in range(0x0900, 0x0966)], + *[chr(c) for c in range(0x0970, 0x0980)], +).optimize() + NEMO_HEX = pynini.union(*string.hexdigits).optimize() NEMO_NON_BREAKING_SPACE = u"\u00a0" NEMO_ZWNJ = u"\u200c" @@ -83,6 +98,34 @@ def generator_main(file_name: str, graphs: Dict[str, 'pynini.FstLike']): logging.info(f'Created {file_name}') +def load_symbols(path): + """ + Builds a dict mapping a symbol name to an FST that deletes a spoken Hindi + phrase and inserts its written form. TSV columns: name, spoken phrase, output + (optional). Rows sharing a name are unioned; "" in the output inserts a + space. + """ + table = {} + with open(path, encoding="utf-8") as f: + for line in f: + line = line.rstrip("\r\n") + if not line or line.startswith("#"): + continue + cols = line.split("\t") + name = cols[0] + words = cols[1].split(" ") + out = cols[2] if len(cols) > 2 else "" + if out == "": + out = " " + fst = pynutil.delete(words[0]) + for word in words[1:]: + fst += delete_space + pynutil.delete(word) + if out: + fst += pynutil.insert(out) + table[name] = (table[name] | fst) if name in table else fst + return table + + def convert_space(fst) -> 'pynini.FstLike': """ Converts space to nonbreaking space. diff --git a/nemo_text_processing/inverse_text_normalization/hi/taggers/electronic.py b/nemo_text_processing/inverse_text_normalization/hi/taggers/electronic.py new file mode 100644 index 000000000..c177c3735 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/taggers/electronic.py @@ -0,0 +1,225 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.inverse_text_normalization.hi.graph_utils import ( + DIGIT_GLYPH_TO_ASCII, + DIGIT_WORD_TO_DEVANAGARI, + GraphFst, + delete_space, + load_symbols, +) +from nemo_text_processing.inverse_text_normalization.hi.utils import get_abs_path +from nemo_text_processing.text_normalization.en.graph_utils import NEMO_ALPHA, TO_LOWER, TO_UPPER + + +class ElectronicFst(GraphFst): + """ + Finite state transducer for classifying electronic expressions in Hindi + inverse text normalization: converts spoken Hindi words into written + electronic forms such as email addresses, URLs, file paths, and domains. + + e-mail: + e.g. कुमार एट जीमेल डॉट कॉम + -> tokens { electronic { username: "kumar" domain: "gmail.com" } } + URL: + e.g. एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश गूगल डॉट कॉम + -> tokens { electronic { domain: "https://google.com" } } + file path (Windows): + e.g. सी कोलन बैकवर्ड स्लैश यूजर्स बैकवर्ड स्लैश एच पी बैकवर्ड स्लैश डेस्कटॉप + -> tokens { electronic { path: "C:\\Users\\HP\\Desktop" } } + file path (Unix/Linux): + e.g. फॉरवर्ड स्लैश होम फॉरवर्ड स्लैश यूजर फॉरवर्ड स्लैश डॉक्युमेंट्स + -> tokens { electronic { path: "/home/user/documents" } } + + """ + + def __init__(self): + super().__init__(name="electronic", kind="classify") + + def seq(atom): + return atom + pynini.closure(delete_space + atom) + + digit_glyphs = DIGIT_GLYPH_TO_ASCII + digit_words = (DIGIT_WORD_TO_DEVANAGARI @ digit_glyphs).optimize() + digit_seq = (digit_glyphs + pynini.closure(digit_glyphs)) | seq(digit_words) + + letter_map_lower = pynini.string_file(get_abs_path("data/electronic/letters.tsv")).invert() + domain_map = pynini.string_file(get_abs_path("data/electronic/domain.tsv")).invert() + server_map = pynini.string_file(get_abs_path("data/electronic/server_name.tsv")).invert() + common_map = pynini.string_file(get_abs_path("data/electronic/common_words.tsv")).invert() + + sym = load_symbols(get_abs_path("data/electronic/symbols.tsv")) + spaced = {name: delete_space + fst + delete_space for name, fst in sym.items()} + + letter_map_upper = (letter_map_lower @ TO_UPPER).optimize() + + to_lower = pynini.closure(TO_LOWER | pynini.project(TO_LOWER, "output")) + common_map_lower = (common_map @ to_lower).optimize() + + latin_run = pynini.closure(NEMO_ALPHA, 1) + latin_run_lower = (latin_run @ to_lower).optimize() + + single_token = server_map | common_map | letter_map_lower + token_seq = seq(single_token) + + path_atom = common_map | server_map | digit_words | digit_glyphs | latin_run | letter_map_upper + path_atom_lower = ( + common_map_lower | server_map | digit_words | digit_glyphs | latin_run_lower | letter_map_lower + ) + unix_path_atom = sym["and"] | path_atom_lower + + file_ext = ( + spaced["dot"] + + seq(path_atom_lower) + + pynini.closure((spaced["dot"] | spaced["hyphen"]) + seq(path_atom_lower)) + ) + + path_sep_seg = (spaced["hyphen"] | spaced["underscore"]) + seq(path_atom) + path_segment = ( + path_atom + + pynini.closure( + (delete_space + path_atom) + | path_sep_seg + | spaced["space"] + | spaced["openbracket"] + | spaced["closebracket"] + ) + + pynini.closure(file_ext, 0, 1) + ) + + unix_sep_seg = (spaced["hyphen"] | spaced["underscore"]) + seq(unix_path_atom) + version_seg = sym["v"] + unix_path_atom + pynini.closure(spaced["dot"] + seq(unix_path_atom)) + dollar_var = spaced["dollar"] + seq(unix_path_atom) + unix_segment = ( + (version_seg | dollar_var | unix_path_atom) + + pynini.closure((delete_space + unix_path_atom) | unix_sep_seg) + + pynini.closure(file_ext, 0, 1) + ) + + lit_seg = ( + unix_path_atom + + pynini.closure((delete_space + unix_path_atom) | unix_sep_seg | spaced["lithyphen"]) + + pynini.closure(file_ext, 0, 1) + ) + + def path_graph(prefix, segment, separator): + return ( + pynutil.insert("path: \"") + + prefix + + segment + + pynini.closure(separator + segment) + + pynini.closure(separator, 0, 1) + + pynutil.insert("\"") + ) + + windows_path_fst = path_graph( + letter_map_upper + delete_space + sym["colon"] + spaced["backslash"], path_segment, spaced["backslash"] + ) + unc_path_fst = path_graph(spaced["backslash"], path_segment, spaced["backslash"]) + unix_abs_path_fst = path_graph(spaced["forwardslash"], unix_segment, spaced["forwardslash"]) + tilde_path_fst = path_graph(sym["tilde"] + spaced["forwardslash"], unix_segment, spaced["forwardslash"]) + unix_rel_path_fst = path_graph(unix_segment + spaced["forwardslash"], unix_segment, spaced["forwardslash"]) + literal_rel_path_fst = path_graph(lit_seg + spaced["litslash"], lit_seg, spaced["litslash"]) + + domain_single = server_map | common_map_lower | letter_map_lower + domain_token_seq = seq(domain_single) + + domain_label = (digit_seq + pynini.closure(delete_space + letter_map_lower)) | domain_token_seq + domain_body = domain_label + pynini.closure(spaced["hyphen"] + domain_label) + compound_tld = domain_map + pynini.closure(spaced["dot"] + domain_map, 0, 2) + full_domain = pynini.closure(domain_body + spaced["dot"], 0, 4) + domain_body + spaced["dot"] + compound_tld + full_domain_bare = pynini.closure(domain_body + spaced["dot"], 0, 4) + domain_body + + uname_atom = sym["and"] | letter_map_lower | digit_words | digit_glyphs | server_map | common_map + uname_sep = spaced["dot"] | spaced["hyphen"] | spaced["underscore"] + username = uname_atom + pynini.closure((uname_sep + uname_atom) | (delete_space + uname_atom)) + email_fst = ( + pynutil.insert("username: \"") + + username + + pynutil.insert("\"") + + spaced["at"] + + pynutil.insert("domain: \"") + + domain_body + + spaced["dot"] + + compound_tld + + pynutil.insert("\"") + ) + + path_atom_url = ( + (digit_seq + spaced["x"] + digit_seq) + | (digit_seq + delete_space + letter_map_lower + delete_space + digit_seq) + | digit_seq + | token_seq + ) + + inline_domain_seg = ( + pynini.closure(token_seq + spaced["dot"], 0, 2) + + token_seq + + spaced["dot"] + + domain_map + + pynini.closure(spaced["dot"] + domain_map, 0, 1) + ) + + path_segment_url = ( + path_atom_url + + pynini.closure(spaced["hyphen"] + (digit_seq | token_seq)) + + pynini.closure(spaced["underscore"] + token_seq) + + pynini.closure(spaced["dot"] + token_seq, 0, 1) + ) + + url_seg = (spaced["dot"] + token_seq) | inline_domain_seg | path_segment_url + www_as_path_seg = sym["www"] + spaced["dot"] + full_domain + pynini.closure(spaced["forwardslash"] + url_seg) + slash_with_word = spaced["forwardslash"] + (url_seg | www_as_path_seg) + + hash_frag = spaced["hashtag"] + token_seq + pynini.closure(spaced["hyphen"] + token_seq) + + url_tail = ( + pynini.closure(slash_with_word) + + pynini.closure(spaced["forwardslash"], 0, 1) + + pynini.closure(hash_frag, 0, 1) + ) + domain_and_path = full_domain + url_tail + domain_and_path_bare = full_domain_bare + url_tail + + protocol_prefix = ( + (sym["https"] | sym["http"]) + delete_space + pynini.closure(sym["www"] + spaced["dot"], 0, 1) + ) + + url_fst = pynutil.insert("domain: \"") + protocol_prefix + domain_and_path + pynutil.insert("\"") + url_fst_bare = pynutil.insert("domain: \"") + protocol_prefix + domain_and_path_bare + pynutil.insert("\"") + www_fst = pynutil.insert("domain: \"") + sym["www"] + spaced["dot"] + domain_and_path + pynutil.insert("\"") + www_fst_bare = ( + pynutil.insert("domain: \"") + sym["www"] + spaced["dot"] + domain_and_path_bare + pynutil.insert("\"") + ) + plain_fst = pynutil.insert("domain: \"") + domain_and_path + pynutil.insert("\"") + + graph = ( + email_fst + | windows_path_fst + | unc_path_fst + | url_fst + | www_fst + | url_fst_bare + | www_fst_bare + | unix_abs_path_fst + | tilde_path_fst + | unix_rel_path_fst + | literal_rel_path_fst + | plain_fst + ) + + self.fst = self.add_tokens(graph).optimize() diff --git a/nemo_text_processing/inverse_text_normalization/hi/taggers/measure.py b/nemo_text_processing/inverse_text_normalization/hi/taggers/measure.py index 59227a436..ba00f64cd 100644 --- a/nemo_text_processing/inverse_text_normalization/hi/taggers/measure.py +++ b/nemo_text_processing/inverse_text_normalization/hi/taggers/measure.py @@ -54,6 +54,26 @@ def __init__(self, cardinal: GraphFst, decimal: GraphFst): measurements_graph = pynini.string_file(get_abs_path("data/measure/measurements.tsv")).invert() paune_graph = pynini.string_file(get_abs_path("data/numbers/paune.tsv")).invert() + math_symbols = pynini.string_file(get_abs_path("data/measure/math_symbols.tsv")) + equal_symbol = pynini.string_file(get_abs_path("data/measure/equal_symbols.tsv")) + + # An operand may be a decimal; without this the math path matches a span + # starting mid decimal and leaks "दशमलव" through as a literal word. + math_number = cardinal_graph + pynini.closure( + delete_space + pynini.cross("दशमलव", ".") + delete_space + decimal.graph, 0, 1 + ) + math_operator = delete_space + math_symbols + delete_space + math_number + math_long_side = math_number + pynini.closure(math_operator, 1) + math_short_side = math_number + pynini.closure(math_operator) + math_operation = math_long_side + delete_space + equal_symbol + delete_space + math_short_side + math_operation |= math_short_side + delete_space + equal_symbol + delete_space + math_long_side + math_graph = ( + pynutil.insert('units: "math" cardinal { integer: "') + + math_operation + + pynutil.insert('" } preserve_order: true') + ) + math_graph = pynutil.add_weight(math_graph, 1.05).optimize() + self.measurements = pynutil.insert("units: \"") + measurements_graph + pynutil.insert("\" ") graph_integer = pynutil.insert("integer_part: \"") + cardinal_graph + pynutil.insert("\"") graph_integer_paune = pynutil.insert("integer_part: \"") + paune_graph + pynutil.insert("\"") @@ -247,6 +267,7 @@ def __init__(self, cardinal: GraphFst, decimal: GraphFst): | graph_exception_bai | address_graph | structured_address_graph + | math_graph ) self.graph = graph.optimize() diff --git a/nemo_text_processing/inverse_text_normalization/hi/taggers/roman.py b/nemo_text_processing/inverse_text_normalization/hi/taggers/roman.py new file mode 100644 index 000000000..84db6da44 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/taggers/roman.py @@ -0,0 +1,94 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.inverse_text_normalization.hi.graph_utils import ( + GraphFst, + delete_space, + insert_space, + integer_to_devanagari, +) +from nemo_text_processing.inverse_text_normalization.hi.utils import get_abs_path, load_labels + + +class RomanFst(GraphFst): + """ + Finite state transducer for classifying spoken numbers as Roman numerals + when they follow a small, fixed set of context key words (chapter, volume, + class numbering). The conversion is deliberately restricted to these + predictable contexts; regnal, papal and product names (e.g. भास्कर-II) are a + documented limitation because the same number is ambiguous between Arabic and + Roman form. Numbers above MAX_NUMBER have no Roman form, so the whole number + falls back to Arabic (Devanagari) digits instead of leaving an in-range prefix + dangling. + e.g. अध्याय तीन -> tokens { roman { key_cardinal: "अध्याय" integer: "III" } } + e.g. कक्षा दस -> tokens { roman { key_cardinal: "कक्षा" integer: "X" } } + e.g. अध्याय चार हजार -> tokens { roman { key_cardinal: "अध्याय" integer: "४०००" } } + + Args: + cardinal: CardinalFst, used to read spoken numbers. + """ + + MAX_NUMBER = 3999 + + def __init__(self, cardinal: GraphFst): + super().__init__(name="roman", kind="classify") + + key_words = [label[0] for label in load_labels(get_abs_path("data/roman/key_words.tsv"))] + key_words_fst = pynini.union(*[pynini.accep(word) for word in key_words]).optimize() + + value_to_roman = { + int(value): roman for roman, value in load_labels(get_abs_path("data/roman/roman_numerals.tsv")) + } + + devanagari_to_roman = pynini.string_map( + [ + (integer_to_devanagari(value), self._int_to_roman(value, value_to_roman)) + for value in range(1, self.MAX_NUMBER + 1) + ] + ).optimize() + + number_to_devanagari = cardinal.graph_no_exception + in_range_to_roman = pynini.compose(number_to_devanagari, devanagari_to_roman).optimize() + + roman_range_devanagari = pynini.determinize( + pynini.project(devanagari_to_roman, "input").rmepsilon() + ).optimize() + all_devanagari = pynini.project(number_to_devanagari, "output").optimize() + above_range_devanagari = pynini.difference(all_devanagari, roman_range_devanagari).optimize() + above_range_to_arabic = pynini.compose(number_to_devanagari, above_range_devanagari).optimize() + + spoken_to_roman = pynini.union(in_range_to_roman, above_range_to_arabic).optimize() + + graph = ( + pynutil.insert('key_cardinal: "') + + key_words_fst + + pynutil.insert('"') + + delete_space + + insert_space + + pynutil.insert('integer: "') + + spoken_to_roman + + pynutil.insert('"') + ) + self.fst = self.add_tokens(graph).optimize() + + def _int_to_roman(self, number, value_to_roman): + roman = "" + for value in sorted(value_to_roman, reverse=True): + while number >= value: + roman += value_to_roman[value] + number -= value + return roman diff --git a/nemo_text_processing/inverse_text_normalization/hi/taggers/serial.py b/nemo_text_processing/inverse_text_normalization/hi/taggers/serial.py new file mode 100644 index 000000000..7a21813a2 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/taggers/serial.py @@ -0,0 +1,98 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.inverse_text_normalization.hi.graph_utils import ( + DEVANAGARI_LETTER, + DIGIT_GLYPH_TO_ASCII, + DIGIT_WORD_TO_DEVANAGARI, + NEMO_SIGMA, + GraphFst, + delete_space, + load_symbols, +) +from nemo_text_processing.inverse_text_normalization.hi.utils import get_abs_path +from nemo_text_processing.text_normalization.en.graph_utils import NEMO_ALPHA, NEMO_DIGIT, TO_UPPER + + +class SerialFst(GraphFst): + """ + Finite state transducer for classifying serial strings, whose segments are + joined by a hyphen (a literal "-" or the spoken word "हाइफ़न"). + e.g. कोविड-उन्नीस -> tokens { name: "कोविड-19" } + e.g. ब्रह्मोस हाइफ़न १ -> tokens { name: "ब्रह्मोस-1" } + e.g. एक-आठ सौ-पाँच सौ पचपन -> tokens { name: "1-800-555" } + e.g. दो स्क्वेर्ड -> tokens { name: "2^2" } + e.g. आई ए तीन दो -> tokens { name: "IA32" } + e.g. जी एस ए टी हाइफ़न एक आठ -> tokens { name: "GSAT-18" } + + Args: + cardinal: CardinalFst, used to read spoken numbers. + """ + + def __init__(self, cardinal: GraphFst): + super().__init__(name="serial", kind="classify") + + cardinal_to_devanagari = cardinal.graph.optimize() + + devanagari_to_ascii = pynini.cdrewrite(DIGIT_GLYPH_TO_ASCII, "", "", NEMO_SIGMA) + spoken_number = pynini.compose(cardinal_to_devanagari | DIGIT_WORD_TO_DEVANAGARI, devanagari_to_ascii) + devanagari_numeral = pynini.closure(DIGIT_GLYPH_TO_ASCII, 1) + number = (spoken_number | devanagari_numeral).optimize() + number_words = pynini.arcmap(pynini.project(number, "input"), map_type="rmweight").optimize() + + devanagari_word = pynini.closure(DEVANAGARI_LETTER, 1) + + letters = pynini.string_file(get_abs_path("data/electronic/letters.tsv")) + letter_names = pynini.project(letters, "output").optimize() + word = pynini.difference(devanagari_word, (number_words | letter_names).optimize()).optimize() + + segment = word | number + + sym = load_symbols(get_abs_path("data/electronic/symbols.tsv")) + word_hyphen = delete_space + sym["hyphen"] + delete_space + delimiter = sym["lithyphen"] | word_hyphen + serial_core = segment + pynini.closure(delimiter + segment, 1) + + power_special = pynini.string_file(get_abs_path("data/serial/power_special.tsv")) + power_prefix = pynini.string_file(get_abs_path("data/serial/power.tsv")) + power_generic = power_prefix + delete_space + number + power_suffix = delete_space + (power_special | power_generic) + power_graph = number + power_suffix + + digit_words = (DIGIT_WORD_TO_DEVANAGARI @ DIGIT_GLYPH_TO_ASCII).optimize() + letter_map_upper = (letters.invert() @ TO_UPPER).optimize() + + alnum_token = DIGIT_GLYPH_TO_ASCII | digit_words | letter_map_upper + alnum_run = alnum_token + pynini.closure(delete_space + alnum_token, 1) + + alnum_hyphen_ext = ( + delete_space + sym["hyphen"] + delete_space + alnum_token + pynini.closure(delete_space + alnum_token) + ) + alnum_point_ext = ( + delete_space + sym["point"] + delete_space + alnum_token + pynini.closure(delete_space + alnum_token) + ) + alnum_body = (alnum_run | (alnum_token + alnum_hyphen_ext)) + pynini.closure( + alnum_hyphen_ext | alnum_point_ext + ) + + contains_alpha = NEMO_SIGMA + NEMO_ALPHA + NEMO_SIGMA + contains_digit = NEMO_SIGMA + NEMO_DIGIT + NEMO_SIGMA + alnum_mix = pynini.intersect(contains_alpha, contains_digit).optimize() + alnum_body = (alnum_body @ alnum_mix).optimize() + + graph = pynutil.insert("name: \"") + (serial_core | power_graph | alnum_body) + pynutil.insert("\"") + self.fst = graph.optimize() diff --git a/nemo_text_processing/inverse_text_normalization/hi/taggers/telephone.py b/nemo_text_processing/inverse_text_normalization/hi/taggers/telephone.py index ad584b58b..3e8d717fe 100644 --- a/nemo_text_processing/inverse_text_normalization/hi/taggers/telephone.py +++ b/nemo_text_processing/inverse_text_normalization/hi/taggers/telephone.py @@ -16,10 +16,13 @@ from pynini.lib import pynutil from nemo_text_processing.inverse_text_normalization.hi.graph_utils import ( + DIGIT_GLYPH_TO_ASCII, NEMO_CHAR, + NEMO_HI_DIGIT, NEMO_WHITE_SPACE, GraphFst, delete_space, + load_symbols, ) from nemo_text_processing.inverse_text_normalization.hi.utils import get_abs_path @@ -33,23 +36,34 @@ ) digit = digit_without_shunya | shunya +# Phone numbers are often spoken in two digit groups, e.g. "इक्यासी" for "८१", +# so a single spoken word can contribute two digits to the number. +digit_pair = pynini.string_file(get_abs_path("data/numbers/teens_and_ties.tsv")).invert() +digit_unit = digit | digit_pair -def get_context(keywords: list): - keywords = pynini.union(*keywords) - # Load Hindi digits from TSV files - hindi_digits = ( - pynini.string_file(get_abs_path("data/numbers/digit.tsv")) - | pynini.string_file(get_abs_path("data/numbers/zero.tsv")) - ).project("output") +def digit_sequence(length, first=None, allow_pairs=True): + """ + Sequence of spoken number words producing exactly `length` digits. + + A word may contribute one digit ("नौ" -> "९") or two ("इक्यासी" -> "८१"), so the + length is constrained on the output side rather than by counting spoken words. + `first` optionally restricts the leading digit, e.g. non zero for mobile numbers. + `allow_pairs` may be cleared to accept only single digit words. + """ + unit = digit_unit if allow_pairs else digit + sequence = pynini.closure(unit + delete_space) + unit + if first is None: + output = pynini.closure(NEMO_HI_DIGIT, length, length) + else: + output = first + pynini.closure(NEMO_HI_DIGIT, length - 1, length - 1) + return pynini.compose(sequence, output).optimize() + - # Load English digits from TSV files - english_digits = ( - pynini.string_file(get_abs_path("data/telephone/eng_digit.tsv")) - | pynini.string_file(get_abs_path("data/telephone/eng_zero.tsv")) - ).project("output") +def get_context(keywords: list): + keywords = pynini.union(*keywords) - all_digits = hindi_digits | english_digits + all_digits = pynini.project(digit, "input") non_digit_char = pynini.difference(NEMO_CHAR, pynini.union(all_digits, NEMO_WHITE_SPACE)) word = pynini.closure(non_digit_char, 1) + NEMO_WHITE_SPACE @@ -60,13 +74,27 @@ def get_context(keywords: list): return before, after +def get_optional_extension(): + """ + Optional telephone extension, e.g. "एक्सटेंशन एक दो तीन" -> " ext. १२३". + + Only reachable after a complete phone number, so place names such as + "ग्रीन पार्क एक्सटेंशन" cannot trigger it. + """ + ext_phrase = pynini.string_file(get_abs_path("data/telephone/extension.tsv")) + ext_digits = digit + pynini.closure(delete_space + digit, 0, 4) + return pynini.closure( + delete_space + pynutil.insert(" ") + ext_phrase + pynutil.insert(" ") + delete_space + ext_digits, 0, 1 + ) + + def generate_context_graph(context_keywords, length): context_before, context_after = get_context(context_keywords) - digits = pynini.closure(digit + delete_space, length - 1, length - 1) + digit + digits = digit_sequence(length) graph_after_context = digits + NEMO_WHITE_SPACE + context_after graph_before_context = context_before + NEMO_WHITE_SPACE + digits - graph_without_context = digits + graph_without_context = digit_sequence(length, allow_pairs=False) return ( pynutil.insert("number_part: \"") @@ -86,7 +114,7 @@ def generate_credit(context_keywords): def generate_mobile(context_keywords): context_before, context_after = get_context(context_keywords) - country_code = pynini.cross("प्लस", "+") + pynini.closure(delete_space + digit, 2, 2) + NEMO_WHITE_SPACE + country_code = pynini.cross("प्लस", "+") + delete_space + digit_sequence(2) + NEMO_WHITE_SPACE graph_country_code = ( pynutil.insert("country_code: \"") + (context_before + NEMO_WHITE_SPACE) ** (0, 1) @@ -94,10 +122,11 @@ def generate_mobile(context_keywords): + pynutil.insert("\" ") ) - number_part = digit_without_shunya + delete_space + pynini.closure(digit + delete_space, 8, 8) + digit + number_part = digit_sequence(10, first=pynini.difference(NEMO_HI_DIGIT, pynini.accep("०"))) graph_number = ( pynutil.insert("number_part: \"") + number_part + + get_optional_extension() + pynini.closure(NEMO_WHITE_SPACE + context_after, 0, 1) + pynutil.insert("\" ") ) @@ -109,13 +138,14 @@ def generate_mobile(context_keywords): def generate_telephone(context_keywords): context_before, context_after = get_context(context_keywords) - landline = shunya + delete_space + pynini.closure(digit + delete_space, 9, 9) + digit + landline = digit_sequence(11, first=pynini.accep("०")) landline_with_context_before = context_before + NEMO_WHITE_SPACE + landline landline_with_context_after = landline + NEMO_WHITE_SPACE + context_after return ( pynutil.insert("number_part: \"") + (landline | landline_with_context_before | landline_with_context_after) + + get_optional_extension() + pynutil.insert("\" ") ) @@ -124,6 +154,8 @@ class TelephoneFst(GraphFst): """ Finite state transducer for classifying telephone numbers, e.g. e.g. प्लस इक्यानवे नौ आठ सात छह पांच चार तीन दो एक शून्य => tokens { name: "+९१ ९८७६५ ४३२१०" } + This class also supports IP addresses, e.g. + e.g. एक नौ दो डॉट एक छह आठ डॉट एक डॉट एक => tokens { telephone { number_part: "192.168.1.1" } } Args: Cardinal: CardinalFst """ @@ -134,26 +166,26 @@ def __init__(self, cardinal: GraphFst): # Load context cues from TSV file context_cues = pynini.string_file(get_abs_path("data/telephone/context_cues.tsv")) - # Extract keywords for each category - mobile_keywords = pynini.compose(pynutil.delete("mobile"), context_cues).project("output").optimize() - - landline_keywords = pynini.compose(pynutil.delete("landline"), context_cues).project("output").optimize() - - pincode_keywords = pynini.compose(pynutil.delete("pincode"), context_cues).project("output").optimize() + def keywords(category): + return pynini.compose(pynutil.delete(category), context_cues).project("output").optimize() - credit_keywords = pynini.compose(pynutil.delete("credit"), context_cues).project("output").optimize() + mobile = generate_mobile([keywords("mobile")]) + landline = generate_telephone([keywords("landline")]) + pincode = generate_pincode([keywords("pincode")]) + credit = generate_credit([keywords("credit")]) - # Convert FSTs to keyword lists for generate_* functions - mobile = generate_mobile([mobile_keywords]) - landline = generate_telephone([landline_keywords]) - pincode = generate_pincode([pincode_keywords]) - credit = generate_credit([credit_keywords]) + sym = load_symbols(get_abs_path("data/electronic/symbols.tsv")) + ip_dot = delete_space + sym["dot"] + delete_space + ip_digit = pynini.compose(digit, DIGIT_GLYPH_TO_ASCII) | DIGIT_GLYPH_TO_ASCII + ip_octet = ip_digit + pynini.closure(delete_space + ip_digit, 0, 2) + ip_graph = pynutil.insert("number_part: \"") + ip_octet + (ip_dot + ip_octet) ** 3 + pynutil.insert("\" ") graph = ( pynutil.add_weight(mobile, 0.7) | pynutil.add_weight(landline, 0.8) | pynutil.add_weight(credit, 0.9) | pynutil.add_weight(pincode, 1) + | pynutil.add_weight(ip_graph, 0.7) ) self.final = graph.optimize() diff --git a/nemo_text_processing/inverse_text_normalization/hi/taggers/time.py b/nemo_text_processing/inverse_text_normalization/hi/taggers/time.py index 942b5022b..743646bcd 100644 --- a/nemo_text_processing/inverse_text_normalization/hi/taggers/time.py +++ b/nemo_text_processing/inverse_text_normalization/hi/taggers/time.py @@ -151,7 +151,15 @@ def __init__(self, cardinal: GraphFst): | ((graph_saade | graph_sava | graph_paune) + pynini.closure(delete_space + delete_baje)) ) + time_zone_graph = pynini.invert(pynini.string_file(get_abs_path("data/time/time_zone.tsv"))) + final_time_zone_optional = pynini.closure( + delete_space + insert_space + pynutil.insert("zone: \"") + time_zone_graph + pynutil.insert("\""), + 0, + 1, + ) + graph = graph_hms | graph_hm | graph_hs | graph_ms | graph_hour | graph_quarterly_measures + graph = graph + final_time_zone_optional self.graph = graph.optimize() final_graph = self.add_tokens(graph) diff --git a/nemo_text_processing/inverse_text_normalization/hi/taggers/tokenize_and_classify.py b/nemo_text_processing/inverse_text_normalization/hi/taggers/tokenize_and_classify.py index 50abab0e5..08e99b4d7 100644 --- a/nemo_text_processing/inverse_text_normalization/hi/taggers/tokenize_and_classify.py +++ b/nemo_text_processing/inverse_text_normalization/hi/taggers/tokenize_and_classify.py @@ -28,11 +28,14 @@ from nemo_text_processing.inverse_text_normalization.hi.taggers.cardinal import CardinalFst from nemo_text_processing.inverse_text_normalization.hi.taggers.date import DateFst from nemo_text_processing.inverse_text_normalization.hi.taggers.decimal import DecimalFst +from nemo_text_processing.inverse_text_normalization.hi.taggers.electronic import ElectronicFst from nemo_text_processing.inverse_text_normalization.hi.taggers.fraction import FractionFst from nemo_text_processing.inverse_text_normalization.hi.taggers.measure import MeasureFst from nemo_text_processing.inverse_text_normalization.hi.taggers.money import MoneyFst from nemo_text_processing.inverse_text_normalization.hi.taggers.ordinal import OrdinalFst from nemo_text_processing.inverse_text_normalization.hi.taggers.punctuation import PunctuationFst +from nemo_text_processing.inverse_text_normalization.hi.taggers.roman import RomanFst +from nemo_text_processing.inverse_text_normalization.hi.taggers.serial import SerialFst from nemo_text_processing.inverse_text_normalization.hi.taggers.telephone import TelephoneFst from nemo_text_processing.inverse_text_normalization.hi.taggers.time import TimeFst from nemo_text_processing.inverse_text_normalization.hi.taggers.whitelist import WhiteListFst @@ -89,6 +92,12 @@ def __init__( money_graph = money.fst telephone = TelephoneFst(cardinal) telephone_graph = telephone.fst + electronic = ElectronicFst() + electronic_graph = electronic.fst + serial = SerialFst(cardinal) + serial_graph = serial.fst + roman = RomanFst(cardinal) + roman_graph = roman.fst punct_graph = PunctuationFst().fst whitelist_graph = WhiteListFst().fst word_graph = WordFst().fst @@ -103,6 +112,9 @@ def __init__( | pynutil.add_weight(measure_graph, 1.1) | pynutil.add_weight(money_graph, 1.1) | pynutil.add_weight(telephone_graph, 1.1) + | pynutil.add_weight(electronic_graph, 1.1) + | pynutil.add_weight(serial_graph, 1.1) + | pynutil.add_weight(roman_graph, 1.1) | pynutil.add_weight(word_graph, 100) | pynutil.add_weight(whitelist_graph, 1.01) ) diff --git a/nemo_text_processing/inverse_text_normalization/hi/verbalizers/electronic.py b/nemo_text_processing/inverse_text_normalization/hi/verbalizers/electronic.py new file mode 100644 index 000000000..69e0e405c --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/verbalizers/electronic.py @@ -0,0 +1,49 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.inverse_text_normalization.hi.graph_utils import NEMO_NOT_QUOTE, GraphFst, delete_space + + +class ElectronicFst(GraphFst): + """ + Finite state transducer for verbalizing electronic + e.g. tokens { electronic { username: "kumar" domain: "gmail.com" } } -> kumar@gmail.com + e.g. tokens { electronic { domain: "https://google.com" } } -> https://google.com + e.g. tokens { electronic { path: "C:\\Users\\HP\\Desktop" } } -> C:\\Users\\HP\\Desktop + """ + + def __init__(self): + super().__init__(name="electronic", kind="verbalize") + + def field_graph(field_name: str) -> pynini.Fst: + return ( + pynutil.delete(f"{field_name}:") + + delete_space + + pynutil.delete("\"") + + pynini.closure(NEMO_NOT_QUOTE, 1) + + pynutil.delete("\"") + ) + + domain_graph = field_graph("domain") + username_graph = field_graph("username") + path_graph = field_graph("path") + + email_graph = username_graph + pynutil.insert("@") + delete_space + domain_graph + + graph = email_graph | path_graph | domain_graph + + self.fst = self.delete_tokens(graph).optimize() diff --git a/nemo_text_processing/inverse_text_normalization/hi/verbalizers/measure.py b/nemo_text_processing/inverse_text_normalization/hi/verbalizers/measure.py index dc8592ebf..af2ca46a0 100644 --- a/nemo_text_processing/inverse_text_normalization/hi/verbalizers/measure.py +++ b/nemo_text_processing/inverse_text_normalization/hi/verbalizers/measure.py @@ -15,6 +15,7 @@ import pynini from pynini.lib import pynutil + from nemo_text_processing.text_normalization.en.graph_utils import NEMO_CHAR, GraphFst, delete_space @@ -39,7 +40,9 @@ def __init__(self, cardinal: GraphFst, decimal: GraphFst): pynutil.delete("units:") + delete_space + pynutil.delete("\"") - + pynini.difference(pynini.closure(NEMO_CHAR - " ", 1), pynini.accep("address")) + + pynini.difference( + pynini.closure(NEMO_CHAR - " ", 1), pynini.union(pynini.accep("address"), pynini.accep("math")) + ) + pynutil.delete("\"") + delete_space ) @@ -80,6 +83,18 @@ def __init__(self, cardinal: GraphFst, decimal: GraphFst): ) graph |= address_graph + # Math verbalizer: units: "math" cardinal { integer: "२ + २ = ४" } preserve_order: true + math_graph = ( + pynutil.delete("units:") + + delete_space + + pynutil.delete("\"math\"") + + delete_space + + graph_cardinal + + delete_space + + pynini.closure(preserve_order) + ) + graph |= math_graph + delete_tokens = self.delete_tokens(graph) self.decimal = graph_decimal self.fst = delete_tokens.optimize() diff --git a/nemo_text_processing/inverse_text_normalization/hi/verbalizers/postprocessor.py b/nemo_text_processing/inverse_text_normalization/hi/verbalizers/postprocessor.py new file mode 100644 index 000000000..7e215d293 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/verbalizers/postprocessor.py @@ -0,0 +1,59 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.inverse_text_normalization.hi.graph_utils import ( + MIN_NEG_WEIGHT, + NEMO_CHAR, + NEMO_PUNCT, + NEMO_SIGMA, + GraphFst, +) + + +class PostProcessor(GraphFst): + ''' + Postprocessing of ITN, now contains: + 1. removal of the space before a punctuation mark, e.g. "TBXQF4138W ." -> "TBXQF4138W." + 2. removal of the space after an opening bracket, e.g. "( AVIC" -> "(AVIC" + ''' + + def __init__( + self, + remove_space_before_punct: bool = False, + remove_space_after_bracket: bool = False, + ): + super().__init__(name="PostProcessor", kind="processor") + + delete_space = pynutil.delete(" ") + graph = pynini.cdrewrite('', '', '', NEMO_SIGMA) + + if remove_space_before_punct: + punct = NEMO_PUNCT | pynini.union("।", "॥") + allow_space_before = pynini.union("(", "{", "<", pynini.escape("["), "-", "&", '"', "'", "`", "+") + no_space_before = pynini.difference(punct, allow_space_before).optimize() + non_punct = pynini.difference(NEMO_CHAR, no_space_before).optimize() + graph @= pynini.closure( + pynini.closure(non_punct) + + pynini.closure(no_space_before | pynutil.add_weight(delete_space + no_space_before, MIN_NEG_WEIGHT)) + + pynini.closure(non_punct) + ).optimize() + + if remove_space_after_bracket: + brackets = pynini.union("(", "{", "<", pynini.escape("[")) + graph @= pynini.cdrewrite(delete_space, brackets, NEMO_SIGMA, NEMO_SIGMA).optimize() + + self.fst = graph.optimize() diff --git a/nemo_text_processing/inverse_text_normalization/hi/verbalizers/roman.py b/nemo_text_processing/inverse_text_normalization/hi/verbalizers/roman.py new file mode 100644 index 000000000..5b53ee903 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/hi/verbalizers/roman.py @@ -0,0 +1,38 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.inverse_text_normalization.hi.graph_utils import ( + NEMO_NOT_QUOTE, + GraphFst, + delete_space, + insert_space, +) + + +class RomanFst(GraphFst): + """ + Finite state transducer for verbalizing Roman numerals + e.g. tokens { roman { key_cardinal: "अध्याय" integer: "III" } } -> अध्याय III + """ + + def __init__(self): + super().__init__(name="roman", kind="verbalize") + key_cardinal = pynutil.delete("key_cardinal: \"") + pynini.closure(NEMO_NOT_QUOTE, 1) + pynutil.delete("\"") + integer = pynutil.delete("integer: \"") + pynini.closure(NEMO_NOT_QUOTE, 1) + pynutil.delete("\"") + graph = key_cardinal + delete_space + insert_space + integer + delete_tokens = self.delete_tokens(graph) + self.fst = delete_tokens.optimize() diff --git a/nemo_text_processing/inverse_text_normalization/hi/verbalizers/time.py b/nemo_text_processing/inverse_text_normalization/hi/verbalizers/time.py index 99820a781..c7710f07f 100644 --- a/nemo_text_processing/inverse_text_normalization/hi/verbalizers/time.py +++ b/nemo_text_processing/inverse_text_normalization/hi/verbalizers/time.py @@ -102,7 +102,17 @@ def __init__(self): + delete_space ) + zone = ( + pynutil.delete("zone:") + + delete_space + + pynutil.delete("\"") + + pynini.closure(NEMO_CHAR - " ", 1) + + pynutil.delete("\"") + ) + optional_zone = pynini.closure(delete_space + insert_space + zone, 0, 1) + graph = graph_hour | graph_hms | graph_hm | graph_hs | graph_ms + graph = graph + optional_zone delete_tokens = self.delete_tokens(graph) self.fst = delete_tokens.optimize() diff --git a/nemo_text_processing/inverse_text_normalization/hi/verbalizers/verbalize.py b/nemo_text_processing/inverse_text_normalization/hi/verbalizers/verbalize.py index f1a6c55a3..d66d10645 100644 --- a/nemo_text_processing/inverse_text_normalization/hi/verbalizers/verbalize.py +++ b/nemo_text_processing/inverse_text_normalization/hi/verbalizers/verbalize.py @@ -17,10 +17,12 @@ from nemo_text_processing.inverse_text_normalization.hi.verbalizers.cardinal import CardinalFst from nemo_text_processing.inverse_text_normalization.hi.verbalizers.date import DateFst from nemo_text_processing.inverse_text_normalization.hi.verbalizers.decimal import DecimalFst +from nemo_text_processing.inverse_text_normalization.hi.verbalizers.electronic import ElectronicFst from nemo_text_processing.inverse_text_normalization.hi.verbalizers.fraction import FractionFst from nemo_text_processing.inverse_text_normalization.hi.verbalizers.measure import MeasureFst from nemo_text_processing.inverse_text_normalization.hi.verbalizers.money import MoneyFst from nemo_text_processing.inverse_text_normalization.hi.verbalizers.ordinal import OrdinalFst +from nemo_text_processing.inverse_text_normalization.hi.verbalizers.roman import RomanFst from nemo_text_processing.inverse_text_normalization.hi.verbalizers.telephone import TelephoneFst from nemo_text_processing.inverse_text_normalization.hi.verbalizers.time import TimeFst from nemo_text_processing.inverse_text_normalization.hi.verbalizers.whitelist import WhiteListFst @@ -48,6 +50,8 @@ def __init__(self): measure_graph = MeasureFst(cardinal, decimal).fst money_graph = MoneyFst(cardinal, decimal).fst telephone_graph = TelephoneFst(cardinal).fst + electronic_graph = ElectronicFst().fst + roman_graph = RomanFst().fst word_graph = WordFst().fst whitelist_graph = WhiteListFst().fst @@ -63,5 +67,7 @@ def __init__(self): | measure_graph | money_graph | telephone_graph + | electronic_graph + | roman_graph ) self.fst = graph diff --git a/nemo_text_processing/inverse_text_normalization/inverse_normalize.py b/nemo_text_processing/inverse_text_normalization/inverse_normalize.py index 9a6fcc64c..800238c30 100644 --- a/nemo_text_processing/inverse_text_normalization/inverse_normalize.py +++ b/nemo_text_processing/inverse_text_normalization/inverse_normalize.py @@ -38,6 +38,8 @@ class InverseNormalizer(Normalizer): overwrite_cache: set to True to overwrite .far files max_number_of_permutations_per_split: a maximum number of permutations which can be generated from input sequence of tokens. + post_process: whether to apply language-specific punctuation post-processing + (currently used by Hindi). On by default. """ def __init__( @@ -48,6 +50,7 @@ def __init__( cache_dir: str = None, overwrite_cache: bool = False, max_number_of_permutations_per_split: int = 729, + post_process: bool = True, ): assert input_case in ["lower_cased", "cased"] @@ -157,6 +160,18 @@ def __init__( self.lang = lang self.max_number_of_permutations_per_split = max_number_of_permutations_per_split + # Optional punctuation post-processor, applied by Normalizer.normalize(). + # Follows the en/vi pattern: a separate FST held on the normalizer and + # gated behind the `post_process` flag, rather than baked into the verbalizer. + self.post_processor = None + if lang == 'hi' and post_process: + from nemo_text_processing.inverse_text_normalization.hi.verbalizers.postprocessor import PostProcessor + + self.post_processor = PostProcessor( + remove_space_before_punct=True, + remove_space_after_bracket=True, + ) + def inverse_normalize_list(self, texts: List[str], verbose=False) -> List[str]: """ NeMo inverse text normalizer diff --git a/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_cardinal.txt b/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_cardinal.txt index 4a7221675..d2cec97cd 100644 --- a/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_cardinal.txt +++ b/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_cardinal.txt @@ -52,3 +52,16 @@ ढाई सौ~२५० साढ़े सोलह सौ~१६५० सवा सोलह सौ~१६२५ +चालीस पचास~४० ५० +साठ सत्तर~६० ७० +पचास पचास~५० ५० +पैंतीस पैंतालीस~३५ ४५ +सत्तर अस्सी~७० ८० +अस्सी नब्बे~८० ९० +नब्बे पचास~९० ५० +चालीस पचास साठ~४० ५० ६० +पचास साठ सत्तर~५० ६० ७० +चालीस पचास साठ सत्तर~४० ५० ६० ७० +वहाँ चालीस पचास लोग थे~वहाँ ४० ५० लोग थे +स्कोर साठ सत्तर रहा~स्कोर ६० ७० रहा +अस्सी नब्बे प्रतिशत~८० ९० प्रतिशत diff --git a/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_date.txt b/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_date.txt index 402361d71..6ad0d4ee9 100644 --- a/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_date.txt +++ b/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_date.txt @@ -39,4 +39,10 @@ अठाहरवीं शताब्दी~१८वीं शताब्दी एक हज़ार एकवीं शताब्दी~१००१वीं शताब्दी एक सौ उन्नीसवां शताब्दी~११९वां शताब्दी -उन्नीस सौ बीस से छब्बीस तक~१९२०-२६ तक \ No newline at end of file +उन्नीस सौ बीस से छब्बीस तक~१९२०-२६ तक +छ: सौ बाईस ईस्वी~६२२ ई. +चार सौ ईसा पूर्व~४०० ई.पू. +चार हज़ार ईसा पूर्व~४००० ई.पू. +सन् तेरह सौ अट्ठानवे ईस्वी~सन् १३९८ ई. +पाँच मई सत्रह सौ नवासी ईस्वी~५ मई, १७८९ ई. +दो सौ बहतर से दो सौ बानवे ईसा पूर्व~२७२-२९२ ई.पू. diff --git a/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_electronic.txt b/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_electronic.txt new file mode 100644 index 000000000..fb2f8308f --- /dev/null +++ b/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_electronic.txt @@ -0,0 +1,26 @@ +ए आर एन ओ एल डी सी ए टी एच वाई एट एम आई एल एल ई आर हाइफ़न सी ए आर आर डॉट कॉम~arnoldcathy@miller-carr.com +एल टी ए ए एक दो एट जीमेल डॉट कॉम~ltaa12@gmail.com +एच आर यू एस एस ई एल एल एट जीमेल डॉट कॉम~hrussell@gmail.com +ए एन एंड एस आई एन जी एच तीन के एट जीमेल डॉट कॉम~anandsingh3k@gmail.com +एफ ओ एक्स ए डी ए एम एट एंड ई आर एस ओ एन डॉट ऑर्ग~foxadam@anderson.org +बी एन ए ए डॉट कॉम~bnaa.com +एस आर वी हाइफ़न आठ चार डॉट जी ए ए वाई के वी ए ए डी डी डॉट नेट~srv-84.gaaykvaadd.net +एल टी हाइफ़न तीन एक डॉट बी ए आर टी ओ एन डॉट सी ओ एन एल ई वाई डॉट कॉम~lt-31.barton.conley.com +डब्ल्यू डब्ल्यू डब्ल्यू डॉट ए एल यू एन आई वी डॉट ऑर्ग फॉरवर्ड स्लैश~www.aluniv.org/ +डब्ल्यू डब्ल्यू डब्ल्यू डॉट बी ई सी एस डॉट ए सी डॉट इन~www.becs.ac.in +डब्ल्यू डब्ल्यू डब्ल्यू डॉट बी आई टी एस हाइफ़न पी आई एल ए एन आई डॉट ए सी डॉट इन~www.bits-pilani.ac.in +एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश ए आर ओ डी डी डी एच ए ए डॉट नेट फॉरवर्ड स्लैश~https://arodddhaa.net/ +एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश डब्ल्यू डब्ल्यू डब्ल्यू डॉट एम एच ए ए आर ए ए जे डॉट कॉम फॉरवर्ड स्लैश~https://www.mhaaraaj.com/ +एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश एम ए आर टी आई एन ई ज़ेड हाइफ़न एम ए आर टी आई एन डॉट कॉम~https://martinez-martin.com +एच टी टी पी कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश एम एच ए ए डी ई वी डॉट ऑर्ग फॉरवर्ड स्लैश अबाउट डॉट एच टी एम एल~http://mhaadev.org/about.html +एच टी टी पी कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश ए टी आई डॉट ई डी यू~http://ati.edu +सी कोलन बैकवर्ड स्लैश यूज़र्स बैकवर्ड स्लैश एच पी बैकवर्ड स्लैश डेस्कटॉप बैकवर्ड स्लैश ऑडियो अंडर स्कोर फ़ाइल अंडर स्कोर दो डॉट एम पी तीन~C:\Users\HP\Desktop\Audio_file_2.mp3 +सी कोलन बैकवर्ड स्लैश यूज़र्स बैकवर्ड स्लैश एच पी बैकवर्ड स्लैश म्यूज़िक बैकवर्ड स्लैश~C:\Users\HP\Music\ +डी कोलन बैकवर्ड स्लैश यूज़र्स बैकवर्ड स्लैश एच पी बैकवर्ड स्लैश डाउनलोड्स बैकवर्ड स्लैश एस ए एम पी एल ई डॉट एक्स एल एस एक्स~D:\Users\HP\Downloads\SAMPLE.xlsx +ए ए एस सी डॉट एन आई सी डॉट इन फॉरवर्ड स्लैश~aasc.nic.in/ +होम / लिब्रे ऑफिस - इंप्रेस - टेम्पलेट्स - मास्टर /~home/libreoffice-impress-templates-master/ +फॉरवर्ड स्लैश होम फॉरवर्ड स्लैश पी आई फॉरवर्ड स्लैश ए एन यू जे~/home/pi/anuj +फॉरवर्ड स्लैश होम फॉरवर्ड स्लैश एम ओ एन ई आर ओ हाइफ़न जी यू आई हाइफ़न वी शून्य डॉट एक दो डॉट तीन डॉट शून्य फॉरवर्ड स्लैश~/home/monero-gui-v0.12.3.0/ +सी कोलन बैकवर्ड स्लैश यूज़र्स बैकवर्ड स्लैश एच पी बैकवर्ड स्लैश डेस्कटॉप बैकवर्ड स्लैश चैप्टर स्पेस एक स्पेस पाइथन डॉट पी डी एफ~C:\Users\HP\Desktop\chapter 1 python.pdf +फॉरवर्ड स्लैश ई टी सी फॉरवर्ड स्लैश सी ए हाइफ़न सी ई आर टी आई एफ आई सी ए टी ई एस डॉट सी ओ एन एफ डॉट डी पी के जी हाइफ़न ओ एल डी~/etc/ca-certificates.conf.dpkg-old +फॉरवर्ड स्लैश वर्क फॉरवर्ड स्लैश ओ एस अंडर स्कोर सी ओ ओ आर डी अंडर स्कोर ओ आर डी आई एन ए एन सी ई अंडर स्कोर सर्वे डॉट एच~/work/os_coord_ordinance_survey.h diff --git a/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_measure.txt b/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_measure.txt index 21615f1c5..79d66804d 100644 --- a/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_measure.txt +++ b/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_measure.txt @@ -46,3 +46,34 @@ दो बाई दो~२x२ पाँच बाई पाँच~५x५ बाईस बाई पाँच घन फीट~२२x५ ft³ +दो प्लस दो इक्वल चार~२+२=४ +सात प्लस तीन इक्वल्स दस~७+३=१० +दो जमा दो बराबर चार~२+२=४ +पंद्रह जोड़ पाँच बराबर बीस~१५+५=२० +पाँच माइनस तीन इक्वल दो~५-३=२ +बीस घटा आठ बराबर बारह~२०-८=१२ +दो मल्टीप्लाइड तीन इक्वल छह~२×३=६ +चार गुणा पाँच बराबर बीस~४×५=२० +नौ इनटू तीन बराबर सत्ताईस~९×३=२७ +छह डिवाइडेड दो इक्वल तीन~६/२=३ +सौ भाग चार बराबर पच्चीस~१००/४=२५ +दो प्लस तीन प्लस चार इक्वल नौ~२+३+४=९ +दस माइनस दो माइनस तीन इक्वल पाँच~१०-२-३=५ +चार इक्वल दो प्लस दो~४=२+२ +सात जोड़ तीन इज़ इक्वल टू दस~७+३=१० +दो गुणा तीन बराबर होता है छह~२×३=६ +दो दशमलव पाँच प्लस एक इक्वल तीन दशमलव पाँच~२.५+१=३.५ +दस दशमलव पाँच माइनस दो इक्वल आठ दशमलव पाँच~१०.५-२=८.५ +दो दशमलव पाँच गुणा दो इक्वल पाँच~२.५×२=५ +दो प्लस एक दशमलव पाँच इक्वल तीन दशमलव पाँच~२+१.५=३.५ +चार इक्वल दो दशमलव पाँच प्लस एक दशमलव पाँच~४=२.५+१.५ +दो प्लस तीन इक्वल एक प्लस चार~२+३=१+४ +दस माइनस दो इक्वल तीन प्लस पाँच~१०-२=३+५ +दो दशमलव पाँच~२.५ +दो दशमलव पाँच प्लस एक~२.५ प्लस १ +दो गुना तीन~२ गुना ३ +कीमत दो गुना बढ़ी~कीमत २ गुना बढ़ी +आबादी तीन गुना हो गई~आबादी ३ गुना हो गई +तीन बराबर तीन~३ बराबर ३ +यह संख्या दो गुना बराबर नहीं है~यह संख्या २ गुना बराबर नहीं है +उसकी उम्र पैंतीस चालीस साल है~उसकी उम्र ३५ ४० yr है diff --git a/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_roman.txt b/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_roman.txt new file mode 100644 index 000000000..bb9b8e62c --- /dev/null +++ b/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_roman.txt @@ -0,0 +1,20 @@ +अध्याय एक~अध्याय I +अध्याय तीन~अध्याय III +अध्याय चार~अध्याय IV +अध्याय नौ~अध्याय IX +खंड पाँच~खंड V +खण्ड सात~खण्ड VII +कक्षा दस~कक्षा X +कक्षा बारह~कक्षा XII +अध्याय बीस~अध्याय XX +अध्याय चालीस~अध्याय XL +अध्याय निन्यानवे~अध्याय XCIX +अध्याय बयालीस~अध्याय XLII +अध्याय चालीस दो~अध्याय XL २ +अध्याय चौदह~अध्याय XIV +अध्याय उन्नीस~अध्याय XIX +अध्याय चार सौ~अध्याय CD +अध्याय सौ~अध्याय C +अध्याय नौ सौ~अध्याय CM +अध्याय एक हजार~अध्याय M +अध्याय चार हजार~अध्याय ४००० diff --git a/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_serial.txt b/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_serial.txt new file mode 100644 index 000000000..a61cb28a9 --- /dev/null +++ b/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_serial.txt @@ -0,0 +1,21 @@ +कोविड-उन्नीस~कोविड-19 +ब्रह्मोस-एक~ब्रह्मोस-1 +निन्यानबे-एक~99-1 +दस-बीस-तीस~10-20-30 +एक-आठ सौ-पाँच सौ पचपन~1-800-555 +दो स्क्वेर्ड~2^2 +दो क्यूब~2^3 +चार टु द पावर पाँच~4^5 +ब्रह्मोस हाइफ़न १~ब्रह्मोस-1 +मंगल हाइफ़न दो~मंगल-2 +आई ए तीन दो~IA32 +आठ सात बी एफ जे यू आठ नौ छह शून्य वी एफ~87BFJU8960VF +आर दो पाँच शून्य एस प्रो~R250S प्रो +एक दो शून्य एम एम~120MM +ऑडी ए छह~ऑडी A6 +ऑपपो एन्को एक्स दो~ऑपपो एन्को X2 +सी छह एच एक दो ओ छह~C6H12O6 +जी एस ए टी हाइफ़न एक आठ~GSAT-18 +डब्ल्यू डब्ल्यू डब्ल्यू सात सात आठ~WWW778 +पी एम दो प्वाइंट पाँच~PM2.5 +टी हाइफ़न एक्स चार शून्य शून्य~T-X400 diff --git a/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_telephone.txt b/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_telephone.txt index 3b84a333d..088afac32 100644 --- a/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_telephone.txt +++ b/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_telephone.txt @@ -25,4 +25,19 @@ zero one three four two three two one five four eight~०१३४२३२१ चार चार चार चार~४४४४ सात आठ नौ एक~७८९१ एक शून्य दो शून्य~१०२० -नौ आठ सात छह~९८७६ \ No newline at end of file +नौ आठ सात छह~९८७६ +दो पाँच पाँच डॉट एक छह आठ डॉट चार छह डॉट एक सात पाँच~255.168.46.175 +दो पाँच डॉट दो तीन तीन डॉट एक चार डॉट दो तीन एक~25.233.14.231 +नौ आठ सात छह पाँच चार तीन दो एक शून्य एक्सटेंशन एक दो तीन~९८७६५४३२१० ext. १२३ +प्लस नौ एक नौ आठ सात छह पाँच चार तीन दो एक शून्य एक्सटेंशन चार पाँच~+९१ ९८७६५४३२१० ext. ४५ +शून्य एक एक दो छह एक दो तीन चार पाँच छह एक्सटेंशन दो दो~०११२६१२३४५६ ext. २२ +नौ आठ सात छह पाँच चार तीन दो एक शून्य एक्स्टेंशन एक दो तीन~९८७६५४३२१० ext. १२३ +नौ आठ सात छह पाँच चार तीन दो एक शून्य एक्सटेंशन नंबर सात आठ~९८७६५४३२१० ext. ७८ +सात शून्य एक दो तीन चार पाँच छह सात आठ एक्सटेंशन नौ~७०१२३४५६७८ ext. ९ +प्लस इक्यानवे इक्यासी दस इकसठ सड़सठ शून्य आठ~+९१ ८११०६१६७०८ +प्लस इक्यानवे अड़सठ चालीस अठासी उनासी अड़तीस~+९१ ६८४०८८७९३८ +इक्यासी दस इकसठ सड़सठ शून्य आठ~८११०६१६७०८ +अड़सठ चालीस अठासी उनासी अड़तीस~६८४०८८७९३८ +नौ आठ सात छह पाँच तैंतालीस इक्कीस शून्य~९८७६५४३२१० +शून्य बीस चौबीस सैंतीस पंद्रह बयालीस~०२०२४३७१५४२ +प्लस इक्यानवे नौ आठ सात छह पाँच चार तीन दो एक शून्य~+९१ ९८७६५४३२१० diff --git a/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_time.txt b/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_time.txt index 8ec5e4df3..244be9a47 100644 --- a/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_time.txt +++ b/tests/nemo_text_processing/hi/data_inverse_text_normalization/test_cases_time.txt @@ -23,3 +23,31 @@ साढ़े ग्यारह~११:३० पौने पाँच~४:४५ पौने तीन घंटा~२:४५ +सुबह के दो बजे~सुबह के २:०० +दोपहर के तीन बजे~दोपहर के ३:०० +शाम के छह बजे~शाम के ६:०० +रात के आठ बजे~रात के ८:०० +सुबह के छह बजकर उनतालीस मिनट~सुबह के ६:३९ +दोपहर के एक बजकर छत्तीस मिनट~दोपहर के १:३६ +शाम के सात बजकर बीस मिनट~शाम के ७:२० +रात के ग्यारह बजकर पचास मिनट~रात के ११:५० +सुबह के पाँच बजकर बीस मिनट बारह सेकंड~सुबह के ५:२०:१२ +दोपहर के तीन बजकर उनसठ मिनट छत्तीस सेकंड~दोपहर के ३:५९:३६ +रात के दस बजकर अड़तालीस मिनट पचास सेकंड~रात के १०:४८:५० +सुबह के पाँच बजके एक मिनट~सुबह के ५:०१ +दोपहर के तीन बजके एक मिनट~दोपहर के ३:०१ +पाँच बजे आई एस टी~५:०० IST +छह बजे जी एम टी~६:०० GMT +दो बजे पी एस टी~२:०० PST +पाँच बजकर तीस मिनट आई एस टी~५:३० IST +नौ बजकर पंद्रह मिनट यू टी सी~९:१५ UTC +ग्यारह बजकर पचास मिनट ई एस टी~११:५० EST +सात बजकर बीस मिनट दस सेकंड आई एस टी~७:२०:१० IST +सुबह के आठ बजे आई एस टी~सुबह के ८:०० IST +रात के दस बजकर तीस मिनट जी एम टी~रात के १०:३० GMT +पाँच बजे भारतीय मानक समय~५:०० IST +पाँच बजे भारतीय समयानुसार~५:०० IST +साढ़े पाँच बजे आई एस टी~५:३० IST +सवा चार बजे जी एम टी~४:१५ GMT +पौने नौ बजे आई एस टी~८:४५ IST +ढाई बजे आई एस टी~२:३० IST diff --git a/tests/nemo_text_processing/hi/test_address.py b/tests/nemo_text_processing/hi/test_address.py index f01dc76c3..41f5360ac 100644 --- a/tests/nemo_text_processing/hi/test_address.py +++ b/tests/nemo_text_processing/hi/test_address.py @@ -39,4 +39,4 @@ def test_norm(self, test_input, expected): @pytest.mark.unit def test_denorm(self, test_input, expected): pred = self.inverse_normalizer.inverse_normalize(test_input, verbose=False) - assert pred.strip() == expected.strip() + assert pred == expected diff --git a/tests/nemo_text_processing/hi/test_date.py b/tests/nemo_text_processing/hi/test_date.py index df12e9874..71e351641 100644 --- a/tests/nemo_text_processing/hi/test_date.py +++ b/tests/nemo_text_processing/hi/test_date.py @@ -39,4 +39,4 @@ def test_norm(self, test_input, expected): @pytest.mark.unit def test_denorm(self, test_input, expected): pred = self.inverse_normalizer.inverse_normalize(test_input, verbose=False) - assert pred.strip() == expected.strip() + assert pred == expected diff --git a/tests/nemo_text_processing/hi/test_decimal.py b/tests/nemo_text_processing/hi/test_decimal.py index 582b59422..368266300 100644 --- a/tests/nemo_text_processing/hi/test_decimal.py +++ b/tests/nemo_text_processing/hi/test_decimal.py @@ -39,4 +39,4 @@ def test_norm(self, test_input, expected): @pytest.mark.unit def test_denorm(self, test_input, expected): pred = self.inverse_normalizer.inverse_normalize(test_input, verbose=False) - assert pred.strip() == expected.strip() + assert pred == expected diff --git a/tests/nemo_text_processing/hi/test_electronic.py b/tests/nemo_text_processing/hi/test_electronic.py index 6b2e3b4f0..9086bd2a9 100644 --- a/tests/nemo_text_processing/hi/test_electronic.py +++ b/tests/nemo_text_processing/hi/test_electronic.py @@ -15,6 +15,7 @@ import pytest from parameterized import parameterized +from nemo_text_processing.inverse_text_normalization.inverse_normalize import InverseNormalizer from nemo_text_processing.text_normalization.normalize import Normalizer from ..utils import CACHE_DIR, parse_test_case_file @@ -24,6 +25,7 @@ class TestElectronic: normalizer = Normalizer( input_case='cased', lang='hi', cache_dir=CACHE_DIR, overwrite_cache=False, post_process=True ) + inverse_normalizer = InverseNormalizer(lang='hi', cache_dir=CACHE_DIR, overwrite_cache=False) @parameterized.expand(parse_test_case_file('hi/data_text_normalization/test_cases_electronic.txt')) @pytest.mark.run_only_on('CPU') @@ -31,3 +33,10 @@ class TestElectronic: def test_norm(self, test_input, expected): pred = self.normalizer.normalize(test_input, verbose=False, punct_post_process=True) assert pred == expected + + @parameterized.expand(parse_test_case_file('hi/data_inverse_text_normalization/test_cases_electronic.txt')) + @pytest.mark.run_only_on('CPU') + @pytest.mark.unit + def test_denorm(self, test_input, expected): + pred = self.inverse_normalizer.inverse_normalize(test_input, verbose=False) + assert pred == expected diff --git a/tests/nemo_text_processing/hi/test_fraction.py b/tests/nemo_text_processing/hi/test_fraction.py index bedf9d0f7..4d063af81 100644 --- a/tests/nemo_text_processing/hi/test_fraction.py +++ b/tests/nemo_text_processing/hi/test_fraction.py @@ -39,4 +39,4 @@ def test_norm(self, test_input, expected): @pytest.mark.unit def test_denorm(self, test_input, expected): pred = self.inverse_normalizer.inverse_normalize(test_input, verbose=False) - assert pred.strip() == expected.strip() + assert pred == expected diff --git a/tests/nemo_text_processing/hi/test_measure.py b/tests/nemo_text_processing/hi/test_measure.py index 71352cdc8..6cf5d930c 100644 --- a/tests/nemo_text_processing/hi/test_measure.py +++ b/tests/nemo_text_processing/hi/test_measure.py @@ -39,4 +39,4 @@ def test_norm(self, test_input, expected): @pytest.mark.unit def test_denorm(self, test_input, expected): pred = self.inverse_normalizer.inverse_normalize(test_input, verbose=False) - assert pred.strip() == expected.strip() + assert pred == expected diff --git a/tests/nemo_text_processing/hi/test_money.py b/tests/nemo_text_processing/hi/test_money.py index 0665146a6..3e3a9b58f 100644 --- a/tests/nemo_text_processing/hi/test_money.py +++ b/tests/nemo_text_processing/hi/test_money.py @@ -39,4 +39,4 @@ def test_norm(self, test_input, expected): @pytest.mark.unit def test_denorm(self, test_input, expected): pred = self.inverse_normalizer.inverse_normalize(test_input, verbose=False) - assert pred.strip() == expected.strip() + assert pred == expected diff --git a/tests/nemo_text_processing/hi/test_roman.py b/tests/nemo_text_processing/hi/test_roman.py index 041b88fd1..3e32c6e8a 100644 --- a/tests/nemo_text_processing/hi/test_roman.py +++ b/tests/nemo_text_processing/hi/test_roman.py @@ -12,11 +12,9 @@ # See the License for the specific language governing permissions and # limitations under the License. - import pytest from parameterized import parameterized -from nemo_text_processing.inverse_text_normalization.inverse_normalize import InverseNormalizer from nemo_text_processing.text_normalization.normalize import Normalizer from ..utils import CACHE_DIR, parse_test_case_file @@ -26,11 +24,10 @@ class TestRoman: normalizer = Normalizer( input_case='cased', lang='hi', cache_dir=CACHE_DIR, overwrite_cache=False, post_process=False ) - inverse_normalizer = InverseNormalizer(lang='hi', cache_dir=CACHE_DIR, overwrite_cache=False) @parameterized.expand(parse_test_case_file('hi/data_text_normalization/test_cases_roman.txt')) @pytest.mark.run_only_on('CPU') @pytest.mark.unit def test_norm(self, test_input, expected): pred = self.normalizer.normalize(test_input, verbose=False) - assert pred.strip() == expected.strip() + assert pred == expected diff --git a/tests/nemo_text_processing/hi/test_sparrowhawk_inverse_text_normalization.sh b/tests/nemo_text_processing/hi/test_sparrowhawk_inverse_text_normalization.sh index 0e31a1a00..7b0aff71d 100644 --- a/tests/nemo_text_processing/hi/test_sparrowhawk_inverse_text_normalization.sh +++ b/tests/nemo_text_processing/hi/test_sparrowhawk_inverse_text_normalization.sh @@ -83,6 +83,21 @@ testITNWhiteList() { runtest $input } +testITNElectronic() { + input=$PROJECT_DIR/hi/data_inverse_text_normalization/test_cases_electronic.txt + runtest $input +} + +testITNSerial() { + input=$PROJECT_DIR/hi/data_inverse_text_normalization/test_cases_serial.txt + runtest $input +} + +testITNRoman() { + input=$PROJECT_DIR/hi/data_inverse_text_normalization/test_cases_roman.txt + runtest $input +} + # Load shUnit2 . $PROJECT_DIR/../shunit2/shunit2 diff --git a/tests/nemo_text_processing/hi/test_telephone.py b/tests/nemo_text_processing/hi/test_telephone.py index 7e43f7e82..f480da156 100644 --- a/tests/nemo_text_processing/hi/test_telephone.py +++ b/tests/nemo_text_processing/hi/test_telephone.py @@ -28,4 +28,4 @@ class TestTelephone: @pytest.mark.unit def test_denorm(self, test_input, expected): pred = self.inverse_normalizer.inverse_normalize(test_input, verbose=False) - assert pred.strip() == expected.strip() + assert pred == expected diff --git a/tests/nemo_text_processing/hi/test_time.py b/tests/nemo_text_processing/hi/test_time.py index 402faf414..9ceb9566d 100644 --- a/tests/nemo_text_processing/hi/test_time.py +++ b/tests/nemo_text_processing/hi/test_time.py @@ -39,4 +39,4 @@ def test_norm(self, test_input, expected): @pytest.mark.unit def test_denorm(self, test_input, expected): pred = self.inverse_normalizer.inverse_normalize(test_input, verbose=False) - assert pred.strip() == expected.strip() + assert pred == expected