From 093c1464d1e2b0ff86a7771f0e17241be9b6f1ff Mon Sep 17 00:00:00 2001 From: Shreyas Pawar Date: Fri, 24 Jul 2026 17:02:53 +0000 Subject: [PATCH 1/2] Hi TN Address, Electronic, Serial, and Cardinal FST Optimizations and Bug Fixes Signed-off-by: Shreyas Pawar --- Jenkinsfile | 2 +- .../hi/data/address/context.tsv | 3 +- .../hi/data/address/en_to_hi_mapping.tsv | 2 - .../hi/data/electronic/common_words.tsv | 62 -------- .../hi/data/electronic/domain.tsv | 48 +++---- .../hi/data/electronic/file_extensions.tsv | 68 ++++----- .../hi/data/electronic/protocols.tsv | 10 +- .../hi/data/electronic/server_name.tsv | 25 ---- .../text_normalization/hi/taggers/cardinal.py | 20 ++- .../hi/taggers/electronic.py | 24 +--- .../text_normalization/hi/taggers/measure.py | 136 ++++++++++++------ .../text_normalization/hi/taggers/serial.py | 34 +++-- .../hi/verbalizers/electronic.py | 76 ++++------ .../test_cases_address.txt | 67 +++++---- .../test_cases_cardinal.txt | 11 ++ .../test_cases_electronic.txt | 106 +++++++------- .../test_cases_money.txt | 18 ++- .../test_cases_roman.txt | 2 +- .../test_cases_serial.txt | 11 +- 19 files changed, 356 insertions(+), 369 deletions(-) delete mode 100644 nemo_text_processing/text_normalization/hi/data/address/en_to_hi_mapping.tsv delete mode 100644 nemo_text_processing/text_normalization/hi/data/electronic/common_words.tsv delete mode 100644 nemo_text_processing/text_normalization/hi/data/electronic/server_name.tsv diff --git a/Jenkinsfile b/Jenkinsfile index a6abeb361..35f703f96 100644 --- a/Jenkinsfile +++ b/Jenkinsfile @@ -28,7 +28,7 @@ pipeline { MR_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/03-12-24-1' JA_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/10-17-24-1' KO_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/04-23-26-0' - HI_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/07-10-26-0' + HI_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/07-24-26-0' DEFAULT_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/06-08-23-0' } stages { diff --git a/nemo_text_processing/text_normalization/hi/data/address/context.tsv b/nemo_text_processing/text_normalization/hi/data/address/context.tsv index 9faadaa3b..d57bfd7d3 100644 --- a/nemo_text_processing/text_normalization/hi/data/address/context.tsv +++ b/nemo_text_processing/text_normalization/hi/data/address/context.tsv @@ -44,5 +44,4 @@ वेस्ट सामने पीछे -वीया -आर डी \ No newline at end of file +वीया \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/address/en_to_hi_mapping.tsv b/nemo_text_processing/text_normalization/hi/data/address/en_to_hi_mapping.tsv deleted file mode 100644 index 15929b547..000000000 --- a/nemo_text_processing/text_normalization/hi/data/address/en_to_hi_mapping.tsv +++ /dev/null @@ -1,2 +0,0 @@ -street स्ट्रीट -southern सदर्न \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/electronic/common_words.tsv b/nemo_text_processing/text_normalization/hi/data/electronic/common_words.tsv deleted file mode 100644 index a9f8e937d..000000000 --- a/nemo_text_processing/text_normalization/hi/data/electronic/common_words.tsv +++ /dev/null @@ -1,62 +0,0 @@ -about अबाउट -blog ब्लॉग -home होम -index इंडेक्स -login लॉगिन -register रजिस्टर -search सर्च -tags टैग्स -category केटेगरी -categories केटेगरीज़ -post पोस्ट -posts पोस्ट्स -page पेज -pages पेजेस -user यूज़र -users यूज़र्स -admin एडमिन -app ऐप -help हेल्प -terms टर्म्स -privacy प्राइवेसी -contact कॉन्टैक्ट -main मेन -explore एक्सप्लोर -wiki विकी -docs डॉक्स -download डाउनलोड -downloads डाउनलोड्स -upload अपलोड -uploads अपलोड्स -photos फ़ोटोज़ -images इमेजेज़ -music म्यूज़िक -video वीडियो -videos वीडियोज़ -desktop डेस्कटॉप -documents डॉक्युमेंट्स -tests टेस्ट्स -test टेस्ट -config कॉन्फ़िग -settings सेटिंग्स -profile प्रोफ़ाइल -account अकाउंट -web वेब -email ई मेल -mobile मोबाइल -phone फोन -phones फोन्स -online ऑनलाइन -domain डोमेन -domains डोमेन्स -data डेटा -file फ़ाइल -files फाइल्स -audio ऑडियो -software सॉफ्टवेयर -homepage होमपेज -content कंटेन्ट -default डिफ़ॉल्ट -world वर्ल्ड -list लिस्ट -license लाइसेंस \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/electronic/domain.tsv b/nemo_text_processing/text_normalization/hi/data/electronic/domain.tsv index bdb1a902d..dccb5dd90 100644 --- a/nemo_text_processing/text_normalization/hi/data/electronic/domain.tsv +++ b/nemo_text_processing/text_normalization/hi/data/electronic/domain.tsv @@ -1,24 +1,24 @@ -com कॉम -org ऑर्ग -net नेट -edu ई डी यू -gov जी ओ वी -in इन -co सी ओ -io आई ओ -ai ए आई -uk यू के -us यू एस -au ए यू -ca सी ए -ac ए सी -res आर ई एस -nic एन आई सी -ernet ई आर नेट -tv टी वी -me एम ई -tech टेक -dev डी ई वी -app ऐप -biz बिज़ -info इन्फो \ No newline at end of file +com +org +net +edu +gov +in +co +io +ai +uk +us +au +ca +ac +res +nic +ernet +tv +me +tech +dev +app +biz +info diff --git a/nemo_text_processing/text_normalization/hi/data/electronic/file_extensions.tsv b/nemo_text_processing/text_normalization/hi/data/electronic/file_extensions.tsv index e768c3838..7283febf5 100644 --- a/nemo_text_processing/text_normalization/hi/data/electronic/file_extensions.tsv +++ b/nemo_text_processing/text_normalization/hi/data/electronic/file_extensions.tsv @@ -1,34 +1,34 @@ -jpg जे पी जी -jpeg जे पी ई जी -png पी एन जी -gif जी आई एफ -pdf पी डी एफ -doc डी ओ सी -docx डी ओ सी एक्स -xls एक्स एल एस -xlsx एक्स एल एस एक्स -ppt पी पी टी -pptx पी पी टी एक्स -csv सी एस वी -txt टी एक्स टी -html एच टी एम एल -xml एक्स एम एल -json जे एस ओ एन -css सी एस एस -js जे एस -py पी वाई -java जावा -cpp सी पी पी -zip ज़िप -rar आर ए आर -tar टी ए आर -mp3 एम पी तीन -mp4 एम पी चार -avi ए वी आई -mkv एम के वी -mov एम ओ वी -wav डब्ल्यू ए वी -svg एस वी जी -apk ए पी के -exe ई एक्स ई -sql एस क्यू एल \ No newline at end of file +jpg +jpeg +png +gif +pdf +doc +docx +xls +xlsx +ppt +pptx +csv +txt +html +xml +json +css +js +py +java +cpp +zip +rar +tar +mp3 +mp4 +avi +mkv +mov +wav +svg +apk +exe +sql \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/electronic/protocols.tsv b/nemo_text_processing/text_normalization/hi/data/electronic/protocols.tsv index 627781003..dd897a4b6 100644 --- a/nemo_text_processing/text_normalization/hi/data/electronic/protocols.tsv +++ b/nemo_text_processing/text_normalization/hi/data/electronic/protocols.tsv @@ -1,5 +1,5 @@ -https एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश -http एच टी टी पी कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश -www डब्ल्यू डब्ल्यू डब्ल्यू डॉट -httpswww एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश डब्ल्यू डब्ल्यू डब्ल्यू डॉट -httpwww एच टी टी पी कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश डब्ल्यू डब्ल्यू डब्ल्यू डॉट \ No newline at end of file +https https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश +http http कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश +www www डॉट +httpswww https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश www डॉट +httpwww http कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश www डॉट diff --git a/nemo_text_processing/text_normalization/hi/data/electronic/server_name.tsv b/nemo_text_processing/text_normalization/hi/data/electronic/server_name.tsv deleted file mode 100644 index d0029ee5c..000000000 --- a/nemo_text_processing/text_normalization/hi/data/electronic/server_name.tsv +++ /dev/null @@ -1,25 +0,0 @@ -gmail जीमेल -yahoo याहू -hotmail हॉटमेल -outlook आउटलुक -live लाइव -google गूगल -microsoft माइक्रोसॉफ्ट -facebook फ़ेसबुक -twitter ट्विटर -instagram इंस्टाग्राम -linkedin लिंक्डइन -youtube यूट्यूब -amazon अमेज़ोन -wikipedia विकिपीडिया -github गिटहब -reddit रेडिट -netflix नेटफ्लिक्स -spotify स्पॉटिफाई -apple एप्पल -samsung सैमसंग -nvidia एनविडिया -intel इंटेल -adobe अडोब -wordpress वर्डप्रेस -blogger ब्लॉगर \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/taggers/cardinal.py b/nemo_text_processing/text_normalization/hi/taggers/cardinal.py index c29ccaa59..96afdf885 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/cardinal.py +++ b/nemo_text_processing/text_normalization/hi/taggers/cardinal.py @@ -348,8 +348,26 @@ def create_larger_number_graph(digit_graph, suffix, zeros_counts, sub_graph): ) cardinal_with_leading_zeros = pynutil.add_weight(cardinal_with_leading_zeros, 0.5) + # Handle large numbers written with digit-group separators. + delete_separator = pynutil.delete(",") + two_digits = NEMO_ALL_DIGIT + NEMO_ALL_DIGIT + three_digits = NEMO_ALL_DIGIT + NEMO_ALL_DIGIT + NEMO_ALL_DIGIT + # Indian grouping: 1-2 leading digits, groups of 2, final group of 3. + indian_grouping = ( + pynini.closure(NEMO_ALL_DIGIT, 1, 2) + + pynini.closure(delete_separator + two_digits) + + delete_separator + + three_digits + ) + # International grouping: 1-3 leading digits, one or more groups of 3. + western_grouping = pynini.closure(NEMO_ALL_DIGIT, 1, 3) + pynini.closure( + delete_separator + three_digits, 1 + ) + strip_separators = (indian_grouping | western_grouping).optimize() + cardinal_with_separators = pynini.compose(strip_separators, graph_without_leading_zeros).optimize() + # Full graph including leading zeros - for standalone cardinal matching - final_graph = graph_without_leading_zeros | cardinal_with_leading_zeros + final_graph = graph_without_leading_zeros | cardinal_with_leading_zeros | cardinal_with_separators optional_minus_graph = pynini.closure(pynutil.insert("negative: ") + pynini.cross("-", "\"true\" "), 0, 1) diff --git a/nemo_text_processing/text_normalization/hi/taggers/electronic.py b/nemo_text_processing/text_normalization/hi/taggers/electronic.py index a309ec089..e1b93835e 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/electronic.py +++ b/nemo_text_processing/text_normalization/hi/taggers/electronic.py @@ -22,7 +22,7 @@ class ElectronicFst(GraphFst): """ Finite state transducer for classifying electronic: as URLs, email addresses, file paths, - IP addresses, domains, chemical formulas, and alphanumeric codes. + IP addresses, domains, and chemical formulas. e.g. kumar@gmail.com -> tokens { electronic { username: "kumar" domain: "gmail.com" } } e.g. https://google.com/ -> tokens { electronic { protocol: "https" domain: "google.com/" } } e.g. C:\\Users\\HP\\Desktop -> tokens { electronic { path: "C:\\Users\\HP\\Desktop" } } @@ -157,22 +157,13 @@ def __init__(self, deterministic: bool = True): unbalanced_trailing = pynini.intersect(no_open, ends_with_close) valid_chemical = pynini.difference(raw_chemical, unbalanced_trailing).optimize() - chemical_formula = pynutil.insert("domain: \"") + valid_chemical + pynutil.insert("\"") + # Recognise a chemical formula only when it uses subscript notation + chem_sigma = pynini.closure(NEMO_ALPHA | NEMO_DIGIT | subscript_digit | chemical_symbols) + contains_subscript = chem_sigma + subscript_digit + chem_sigma + valid_chemical = pynini.intersect(valid_chemical, contains_subscript).optimize() - alnum_seg = pynini.closure(NEMO_ALPHA | NEMO_DIGIT, 1) - separator = pynini.accep("-") | pynini.accep(".") - alphanumeric_pattern = alnum_seg + pynini.closure(separator + alnum_seg) - - alnum_hyp_dot_sigma = pynini.closure(NEMO_ALPHA | NEMO_DIGIT | pynini.accep("-") | pynini.accep(".")) - - contains_alpha = alnum_hyp_dot_sigma + NEMO_ALPHA + alnum_hyp_dot_sigma - contains_digit = alnum_hyp_dot_sigma + NEMO_DIGIT + alnum_hyp_dot_sigma - - alphanumeric_code_fst = pynini.intersect( - pynini.intersect(alphanumeric_pattern, contains_alpha), contains_digit - ).optimize() - - alphanumeric_code = pynutil.insert("domain: \"") + alphanumeric_code_fst + pynutil.insert("\"") + # Chemical formulas carry a dedicated tag so the verbalizer can spell element + chemical_formula = pynutil.insert("fragment_id: \"") + valid_chemical + pynutil.insert("\"") graph = ( pynutil.add_weight(url_graph, 1.0) @@ -184,7 +175,6 @@ def __init__(self, deterministic: bool = True): | pynutil.add_weight(combined_domain, 1.1) | pynutil.add_weight(file_with_extension, 1.1) | pynutil.add_weight(chemical_formula, 1.2) - | pynutil.add_weight(alphanumeric_code, 1.2) ) self.graph = graph.optimize() diff --git a/nemo_text_processing/text_normalization/hi/taggers/measure.py b/nemo_text_processing/text_normalization/hi/taggers/measure.py index e18111696..563c1e700 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/measure.py +++ b/nemo_text_processing/text_normalization/hi/taggers/measure.py @@ -28,12 +28,12 @@ HI_SADHE, HI_SAVVA, HYPHEN, - INPUT_LOWER_CASED, LOWERCASE_X, NEMO_CHAR, NEMO_DIGIT, NEMO_HI_DIGIT, NEMO_NOT_SPACE, + NEMO_SIGMA, NEMO_SPACE, NEMO_WHITE_SPACE, ONE_POINT_FIVE, @@ -50,11 +50,18 @@ from nemo_text_processing.text_normalization.hi.utils import get_abs_path digit = pynini.string_file(get_abs_path("data/numbers/digit.tsv")) -# Load both Hindi (Devanagari) and English (Arabic) number mappings teens_ties_hi = pynini.string_file(get_abs_path("data/numbers/teens_and_ties.tsv")) teens_ties_en = pynini.string_file(get_abs_path("data/numbers/teens_and_ties_en.tsv")) teens_ties = pynini.union(teens_ties_hi, teens_ties_en) teens_and_ties = pynutil.add_weight(teens_ties, -0.1) +# Shared Address Maps +zero = pynini.string_file(get_abs_path("data/numbers/zero.tsv")) +telephone_number = pynini.string_file(get_abs_path("data/telephone/number.tsv")) +states_map = pynini.string_file(get_abs_path("data/address/states.tsv")) +cities_map = pynini.string_file(get_abs_path("data/address/cities.tsv")) +special_characters_map = pynini.string_file(get_abs_path("data/address/special_characters.tsv")) +letters_map = pynini.string_file(get_abs_path("data/address/letters.tsv")) +context_map = pynini.string_file(get_abs_path("data/address/context.tsv")) class MeasureFst(GraphFst): @@ -71,32 +78,31 @@ class MeasureFst(GraphFst): for False multiple transduction are generated (used for audio-based normalization) """ - def get_structured_address_graph(self, ordinal: GraphFst, input_case: str): + def get_structured_address_graph(self, cardinal: GraphFst, ordinal: GraphFst, input_case: str): """ Minimal address tagger for state/city + pincode patterns only. - Highly optimized for performance. Examples: "मुंबई ८८४४०४" -> "मुंबई आठ आठ चार चार शून्य चार" "गोवा १२३४५६" -> "गोवा एक दो तीन चार पाँच छह" + "100 फीट रोड, चेन्नई" -> "एक सौ फीट रोड, चेन्नई" """ # State/city keywords - states = pynini.string_file(get_abs_path("data/address/states.tsv")) - cities = pynini.string_file(get_abs_path("data/address/cities.tsv")) - state_city_names = pynini.union(states, cities).optimize() - # Digit mappings - num_token = ( - digit - | pynini.string_file(get_abs_path("data/numbers/zero.tsv")) - | pynini.string_file(get_abs_path("data/telephone/number.tsv")) - ).optimize() + state_city_names = pynini.union(states_map, cities_map).optimize() + + # Digit mappings (shared maps loaded once at module level) + num_token = (digit | zero | telephone_number).optimize() - # Pincode (6 digits) + # Pincode (6 digits) -> always digit-by-digit (length >= 4) pincode = (num_token + pynini.closure(insert_space + num_token, 5, 5)).optimize() - # Street number (1-4 digits) - street_num = (num_token + pynini.closure(insert_space + num_token, 0, 3)).optimize() + # Street number: 1-3 digits read as cardinal, 4+ digits read digit-by-digit + num_1to3 = ( + cardinal.digit | cardinal.zero | cardinal.teens_and_ties | cardinal.graph_hundreds + ).optimize() + num_4plus = (num_token + pynini.closure(insert_space + num_token, 3)).optimize() + street_num = (num_1to3 | num_4plus).optimize() # Text: words with trailing separator (comma? + space) any_digit = pynini.union(NEMO_HI_DIGIT, NEMO_DIGIT).optimize() @@ -107,7 +113,9 @@ def get_structured_address_graph(self, ordinal: GraphFst, input_case: str): # Separator: optional comma followed by mandatory space sep = pynini.closure(pynini.accep(COMMA), 0, 1) + pynini.accep(NEMO_SPACE) word_with_sep = word + sep - text = pynini.closure(word_with_sep, 0, 5).optimize() + # Consume inline address numbers using the same 1-3 (cardinal) / 4+ (digit-by-digit) rule + num_with_sep = street_num + sep + text = pynini.closure(pynini.union(word_with_sep, num_with_sep), 0, 5).optimize() # Pattern: [street_num + sep]? text state/city [space pincode] pattern = ( @@ -124,34 +132,31 @@ def get_structured_address_graph(self, ordinal: GraphFst, input_case: str): ) return pynutil.add_weight(graph, 1.0).optimize() - def get_address_graph(self, ordinal: GraphFst, input_case: str): + def get_address_graph(self, cardinal: GraphFst, ordinal: GraphFst, input_case: str): """ - Address tagger that converts digits/hyphens/slashes character-by-character - when address context keywords are present. - English words and ordinals are converted to Hindi transliterations. - + Address tagger that fires when address context keywords are present. + Examples: - "७०० ओक स्ट्रीट" -> "सात शून्य शून्य ओक स्ट्रीट" - "६६-४ पार्क रोड" -> "छह छह हाइफ़न चार पार्क रोड" + "७०० ओक स्ट्रीट" -> "सात सौ ओक स्ट्रीट" + "६६-४ पार्क रोड" -> "छियासठ हाइफ़न चार पार्क रोड" + "593988" (6-digit pincode) -> "पाँच नौ तीन नौ आठ आठ" + "32A नाज़ प्लाज़ा" -> "बत्तीस ए नाज़ प्लाज़ा" """ - ordinal_graph = ordinal.graph + # Strip internal weights from ordinal graph so a small outer weight suffices + ordinal_graph = pynini.arcmap(ordinal.graph, map_type="rmweight").optimize() # Alphanumeric to word mappings (digits, special characters, telephone digits) char_to_word = ( digit - | pynini.string_file(get_abs_path("data/numbers/zero.tsv")) - | pynini.string_file(get_abs_path("data/address/special_characters.tsv")) - | pynini.string_file(get_abs_path("data/telephone/number.tsv")) + | zero + | special_characters_map + | telephone_number + ).optimize() + letter_to_word = capitalized_input_graph(letters_map) + # Identity acceptor for keywords (Devanagari/English) to prevent unintended rewrites/transliteration + address_keywords = pynini.project( + capitalized_input_graph(context_map), + "input", ).optimize() - letter_to_word = pynini.string_file(get_abs_path("data/address/letters.tsv")) - letter_to_word = capitalized_input_graph(letter_to_word) - address_keywords_hi = pynini.string_file(get_abs_path("data/address/context.tsv")) - - # English address keywords with Hindi translation (case-insensitive) - en_to_hi_map = pynini.string_file(get_abs_path("data/address/en_to_hi_mapping.tsv")) - if input_case != INPUT_LOWER_CASED: - en_to_hi_map = capitalized_input_graph(en_to_hi_map) - address_keywords_en = pynini.project(en_to_hi_map, "input") - address_keywords = pynini.union(address_keywords_hi, address_keywords_en) # Alphanumeric processing: treat digits, letters, and -/ as convertible tokens single_digit = pynini.union(NEMO_DIGIT, NEMO_HI_DIGIT).optimize() @@ -162,20 +167,61 @@ def get_address_graph(self, ordinal: GraphFst, input_case: str): NEMO_CHAR, pynini.union(NEMO_WHITE_SPACE, convertible_char, pynini.accep(COMMA)) ).optimize() - # Token processors with weights: prefer ordinals and known English→Hindi words + # Token processors with weights: prefer ordinals; English words are left untouched # Delete space before comma to avoid Sparrowhawk "sil" issue comma_processor = pynutil.add_weight(delete_space + pynini.accep(COMMA), 0.0) - ordinal_processor = pynutil.add_weight(insert_space + ordinal_graph, -5.0) - english_word_processor = pynutil.add_weight(insert_space + en_to_hi_map, -3.0) + # Slight preference (-0.5) to ensure ordinals beat alphanumeric code splits + ordinal_processor = pynutil.add_weight(insert_space + ordinal_graph, -0.5) + # Pass English words (2+ letters) through unchanged; single letters go to letter_processor + latin_word = single_letter + pynini.closure(single_letter, 1) + english_word_processor = pynutil.add_weight(insert_space + latin_word, 0.1) letter_processor = pynutil.add_weight(insert_space + pynini.compose(single_letter, letter_to_word), 0.5) - digit_char_processor = pynutil.add_weight(insert_space + pynini.compose(convertible_char, char_to_word), 0.0) + # Transliterate separators ("-"/"/"); digits are handled separately by number_run_processor + special_char_processor = pynutil.add_weight(insert_space + pynini.compose(special_chars, char_to_word), 0.0) other_word_processor = pynutil.add_weight(insert_space + pynini.closure(non_space_char, 1), 0.1) + # --- Alphanumeric codes (letter+digit): letters -> Devanagari, digits via cross-class rule --- + code_letter = pynini.compose(single_letter, letter_to_word).optimize() + code_letters = code_letter + pynini.closure(insert_space + code_letter) + code_num_1to3 = ( + cardinal.digit | cardinal.zero | cardinal.teens_and_ties | cardinal.graph_hundreds + ).optimize() + code_num_4plus = pynini.compose( + single_digit ** 4 + pynini.closure(single_digit), cardinal.single_digits_graph + ).optimize() + code_num = (code_num_1to3 | code_num_4plus).optimize() + code_seg = (code_letters | code_num).optimize() + code_delim = pynini.union(HYPHEN, SLASH, NEMO_SPACE).optimize() + code_core = (code_seg + pynini.closure(code_delim + code_seg, 1)).optimize() + # Insert spaces at letter<->digit boundaries for glued codes (e.g., "F16" -> "F 16") + insert_sp_alpha_digit = pynini.cdrewrite(insert_space, single_letter, single_digit, NEMO_SIGMA) + insert_sp_digit_alpha = pynini.cdrewrite(insert_space, single_digit, single_letter, NEMO_SIGMA) + code_space_inserter = pynini.compose(insert_sp_alpha_digit, insert_sp_digit_alpha).optimize() + glued_code = pynini.compose(code_space_inserter, code_core).optimize() + code_transliterate = pynini.union(code_core, glued_code).optimize() + # Restrict to tokens containing BOTH a letter and a digit to prevent fragmented parsing + code_char = pynini.union(single_letter, single_digit, pynini.accep(HYPHEN), pynini.accep(SLASH)) + has_latin_letter = pynini.closure(code_char) + single_letter + pynini.closure(code_char) + has_digit = pynini.closure(code_char) + single_digit + pynini.closure(code_char) + code_only = pynini.intersect( + pynini.intersect(pynini.closure(code_char, 1), has_latin_letter), has_digit + ).optimize() + code_transliterate = pynini.compose(code_only, code_transliterate).optimize() + # Strip internal weights; standalone codes naturally beat split alternatives without outer weights + code_transliterate = pynini.arcmap(code_transliterate, map_type="rmweight").optimize() + code_processor = insert_space + code_transliterate + + # Pure numeric runs: 1-3 digits read as cardinal, 4+ digits read digit-by-digit + number_run = pynini.compose(pynini.closure(single_digit, 1), code_num).optimize() + number_run_processor = pynutil.add_weight(insert_space + number_run, 0.5) + token_processor = ( ordinal_processor | english_word_processor + | code_processor + | number_run_processor | letter_processor - | digit_char_processor + | special_char_processor | pynini.accep(NEMO_SPACE) | comma_processor | other_word_processor @@ -451,8 +497,8 @@ def __init__(self, cardinal: GraphFst, decimal: GraphFst, ordinal: GraphFst, inp + pynutil.insert("\"") ) - address_graph = self.get_address_graph(ordinal, input_case) - structured_address_graph = self.get_structured_address_graph(ordinal, input_case) + address_graph = self.get_address_graph(cardinal, ordinal, input_case) + structured_address_graph = self.get_structured_address_graph(cardinal, ordinal, input_case) graph = ( pynutil.add_weight(graph_decimal, 0.1) diff --git a/nemo_text_processing/text_normalization/hi/taggers/serial.py b/nemo_text_processing/text_normalization/hi/taggers/serial.py index d7433e583..77578ec4c 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/serial.py +++ b/nemo_text_processing/text_normalization/hi/taggers/serial.py @@ -20,6 +20,7 @@ NEMO_DIGIT, NEMO_NOT_SPACE, NEMO_SIGMA, + TO_LOWER, GraphFst, convert_space, ) @@ -38,9 +39,9 @@ class SerialFst(GraphFst): e.g. 2^2 -> tokens { name: "दो स्क्वेर्ड" } e.g. 2^4 -> tokens { name: "दो टु द पावर चार" } e.g. 1-800-555 -> tokens { name: "एक-आठ सौ-पाँच सौ पचपन" } - - Note: Pure Latin-alpha + digit patterns (A12, B-60) are intentionally - excluded here so they fall through to the electronic classifier. + e.g. B-60 -> tokens { name: "बी-साठ" } + e.g. A12 -> tokens { name: "ए बारह" } + e.g. FY2024 -> tokens { name: "एफ वाई दो शून्य दो चार" } """ def __init__( @@ -63,13 +64,19 @@ def __init__( limited_cardinal_graph = ( cardinal.digit | cardinal.zero | cardinal.teens_and_ties | cardinal.graph_hundreds ).optimize() - num_graph = limited_cardinal_graph + + # Number-group sizing for codes: 1-3 digit groups are read as cardinals and 4+ digits are read as digit by digit + digitwise_4plus = pynini.compose( + any_digit ** 4 + pynini.closure(any_digit), cardinal.single_digits_graph + ).optimize() + num_graph = (limited_cardinal_graph | digitwise_4plus).optimize() symbols_graph = pynini.string_file(get_abs_path("data/serial/special_symbols.tsv")).optimize() devanagari_chars = pynini.string_file(get_abs_path("data/serial/chars.tsv")).optimize() - + letter_graph = pynini.string_file(get_abs_path("data/address/letters.tsv")) + letter_graph = (letter_graph | pynini.compose(TO_LOWER, letter_graph)).optimize() latin_letters = letter_graph + pynini.closure(pynutil.insert(" ") + letter_graph) latin_letters = latin_letters.optimize() @@ -110,15 +117,14 @@ def __init__( pure_word_slash = pynini.closure(NEMO_ALPHA, 1) + pynini.accep("/") + pynini.closure(NEMO_ALPHA, 1) + letter_join_char = NEMO_ALPHA | pynini.accep("-") | pynini.accep("/") + contains_latin_letter = pynini.closure(letter_join_char) + NEMO_ALPHA + pynini.closure(letter_join_char) + pure_latin_word = pynini.intersect(pynini.closure(letter_join_char, 1), contains_latin_letter).optimize() + dimension_pattern = ( pynini.closure(any_digit, 1) + (pynini.accep("x") | pynini.accep("X")) + pynini.closure(any_digit, 1) ) - _opt_delim = pynini.closure(pynini.accep("-") | pynini.accep(" "), 0, 1) - latin_alphanum = (pynini.closure(NEMO_ALPHA, 1) + _opt_delim + pynini.closure(any_digit, 1)) | ( - pynini.closure(any_digit, 1) + _opt_delim + pynini.closure(NEMO_ALPHA, 1) - ) - ordinal_suffixes = pynini.project( pynini.union( pynini.string_file(get_abs_path("data/ordinal/suffixes.tsv")), @@ -143,7 +149,13 @@ def __init__( + pynini.union(date_year_suffix, date_suffixes) ) - exclusions = pure_word_slash | dimension_pattern | latin_alphanum | ordinal_pattern | date_pattern + exclusions = ( + pure_word_slash + | pure_latin_word + | dimension_pattern + | ordinal_pattern + | date_pattern + ) accepted_inputs = pynini.difference(NEMO_SIGMA, exclusions).optimize() serial_graph = pynini.compose(accepted_inputs, serial_graph).optimize() diff --git a/nemo_text_processing/text_normalization/hi/verbalizers/electronic.py b/nemo_text_processing/text_normalization/hi/verbalizers/electronic.py index 398c79fef..debd789d4 100644 --- a/nemo_text_processing/text_normalization/hi/verbalizers/electronic.py +++ b/nemo_text_processing/text_normalization/hi/verbalizers/electronic.py @@ -16,6 +16,7 @@ from pynini.lib import pynutil from nemo_text_processing.text_normalization.hi.graph_utils import ( + NEMO_ALPHA, GraphFst, capitalized_input_graph, delete_space, @@ -27,13 +28,15 @@ class ElectronicFst(GraphFst): """ Finite state transducer for verbalizing electronic addresses. - Uses a phonetic-first approach with letter-by-letter fallback. + English words and letters are kept verbatim (Latin script); only digits and + symbols are read out in Hindi. Examples: - electronic { username: "kumar" domain: "gmail.com" } -> "के यू एम ए आर एट जीमेल डॉट कॉम" - electronic { protocol: "https" domain: "google.com/" } -> "एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश गूगल डॉट कॉम फॉरवर्ड स्लैश" - electronic { path: "C:\\Users\\HP\\Desktop" } -> "सी कोलन बैकवर्ड स्लैश यूज़र्स बैकवर्ड स्लैश एच पी बैकवर्ड स्लैश डेस्कटॉप" + electronic { username: "kumar" domain: "gmail.com" } -> "kumar एट gmail डॉट com" + electronic { protocol: "https" domain: "google.com/" } -> "https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश google डॉट com फॉरवर्ड स्लैश" + electronic { path: "C:\\Users\\HP\\Desktop" } -> "C कोलन बैकवर्ड स्लैश Users बैकवर्ड स्लैश HP बैकवर्ड स्लैश Desktop" electronic { domain: "192.168.1.1" } -> "एक नौ दो डॉट एक छह आठ डॉट एक डॉट एक" + electronic { chem: "C₂H₄" } -> "सी दो एच चार" Args: deterministic: if True will provide a single transduction option, @@ -44,11 +47,6 @@ def __init__(self, deterministic: bool = True): super().__init__(name="electronic", kind="verbalize", deterministic=deterministic) symbols_graph = pynini.string_file(get_abs_path("data/electronic/symbols.tsv")).optimize() - domain_graph = pynini.string_file(get_abs_path("data/electronic/domain.tsv")).optimize() - server_name_graph = pynini.string_file(get_abs_path("data/electronic/server_name.tsv")).optimize() - common_words_graph = pynini.string_file(get_abs_path("data/electronic/common_words.tsv")).optimize() - latin_to_hindi_graph = pynini.string_file(get_abs_path("data/address/letters.tsv")) - latin_to_hindi_graph = capitalized_input_graph(latin_to_hindi_graph).optimize() ascii_digit_graph = pynini.string_file(get_abs_path("data/telephone/number.tsv")).optimize() hindi_digit_graph = pynini.string_file(get_abs_path("data/numbers/digit.tsv")).optimize() @@ -58,25 +56,30 @@ def __init__(self, deterministic: bool = True): protocol_graph = pynini.string_file(get_abs_path("data/electronic/protocols.tsv")).optimize() - single_letter = latin_to_hindi_graph + insert_space single_digit = digit_verbalization + insert_space single_symbol = symbols_graph + insert_space single_non_alpha = pynutil.add_weight(single_symbol, 1.0) | pynutil.add_weight(single_digit, 1.0) - def make_alpha_run_verbalizer(tsv_graphs): - phonetic = pynini.union(*[pynutil.add_weight(g + insert_space, w) for g, w in tsv_graphs]) - literal = pynutil.add_weight(pynini.closure(single_letter, 1), 1.1) - return phonetic | literal + # A run of Latin letters is preserved verbatim; digits and symbols verbalize in Hindi. + alpha_run = pynini.closure(NEMO_ALPHA, 1) + insert_space - def make_content(alpha_run_verb, non_alpha_sep=None): + # Chemical formulas are spelled out letter-by-letter (element symbols are + # abbreviations, not words), while digits and symbols verbalize in Hindi. + latin_to_hindi_graph = capitalized_input_graph( + pynini.string_file(get_abs_path("data/address/letters.tsv")) + ).optimize() + chem_char = (latin_to_hindi_graph + insert_space) | single_digit | single_symbol + chem_content = pynini.closure(chem_char, 1) + + def make_content(non_alpha_sep=None): if non_alpha_sep is None: non_alpha_sep = single_non_alpha mandatory_sep = pynini.closure(non_alpha_sep, 1) return ( pynini.closure(non_alpha_sep, 0) - + pynini.closure(alpha_run_verb + mandatory_sep, 0) - + pynini.closure(alpha_run_verb, 0, 1) + + pynini.closure(alpha_run + mandatory_sep, 0) + + pynini.closure(alpha_run, 0, 1) + pynini.closure(non_alpha_sep, 0) ) @@ -84,49 +87,23 @@ def make_content(alpha_run_verb, non_alpha_sep=None): delete_domain_tag = pynutil.delete("domain: \"") delete_protocol_tag = pynutil.delete("protocol: \"") delete_path_tag = pynutil.delete("path: \"") + delete_fragment_id_tag = pynutil.delete("fragment_id: \"") delete_quote = pynutil.delete("\"") - username_alpha_run = make_alpha_run_verbalizer( - [ - (server_name_graph, 0.85), - (domain_graph, 0.87), - (common_words_graph, 0.90), - ] - ) - username_content = make_content(username_alpha_run) + username_content = make_content() username_graph = delete_username_tag + username_content + delete_quote + delete_space + pynutil.insert("एट ") - domain_alpha_run = make_alpha_run_verbalizer( - [ - (server_name_graph, 0.85), - (domain_graph, 0.87), - (common_words_graph, 0.90), - ] - ) - - domain_alpha_run = make_alpha_run_verbalizer( - [ - (server_name_graph, 0.85), - (domain_graph, 0.87), - (common_words_graph, 0.90), - ] - ) - - domain_content = pynutil.add_weight(make_content(domain_alpha_run), 1.0) + domain_content = pynutil.add_weight(make_content(), 1.0) domain_only_graph = delete_domain_tag + domain_content + delete_quote protocol_only_graph = delete_protocol_tag + protocol_graph + insert_space + delete_quote + delete_space - path_alpha_run = make_alpha_run_verbalizer( - [ - (domain_graph, 0.87), - (common_words_graph, 0.90), - ] - ) - path_content = make_content(path_alpha_run) + path_content = make_content() path_graph = delete_path_tag + path_content + delete_quote + chem_graph = delete_fragment_id_tag + chem_content + delete_quote + ip_char = single_symbol | single_digit ip_content = pynini.closure(ip_char, 1) ip_graph = delete_domain_tag + ip_content + delete_quote @@ -140,6 +117,7 @@ def make_content(alpha_run_verb, non_alpha_sep=None): | pynutil.add_weight(path_graph, 1.02) | pynutil.add_weight(ip_graph, 1.03) | pynutil.add_weight(domain_only_graph, 1.04) + | pynutil.add_weight(chem_graph, 1.04) ) delete_tokens = self.delete_tokens(graph) diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_address.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_address.txt index 9989fa75c..d0554ce30 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_address.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_address.txt @@ -1,47 +1,44 @@ -700 ओक स्ट्रीट~सात शून्य शून्य ओक स्ट्रीट -११ जंगल रोड~एक एक जंगल रोड -301 पार्क एवेन्यू~तीन शून्य एक पार्क एवेन्यू -गली नंबर १७ जीएकगढ़~गली नंबर एक सात जीएकगढ़ -अदनान अपार्टमेंट फ्लैट नंबर 55~अदनान अपार्टमेंट फ्लैट नंबर पाँच पाँच +700 ओक स्ट्रीट~सात सौ ओक स्ट्रीट +११ जंगल रोड~ग्यारह जंगल रोड +301 पार्क एवेन्यू~तीन सौ एक पार्क एवेन्यू +गली नंबर १७ जीएकगढ़~गली नंबर सत्रह जीएकगढ़ +अदनान अपार्टमेंट फ्लैट नंबर 55~अदनान अपार्टमेंट फ्लैट नंबर पचपन प्लॉट नंबर ८ बालाजी मार्केट~प्लॉट नंबर आठ बालाजी मार्केट -शॉप नंबर 109 9 और 10 डिवाइडिंग रोड सेक्टर 10 फरीदाबाद~शॉप नंबर एक शून्य नौ नौ और एक शून्य डिवाइडिंग रोड सेक्टर एक शून्य फरीदाबाद -बूथ ७०, सेक्टर ८, चंडीगढ़~बूथ सात शून्य, सेक्टर आठ, चंडीगढ़ -2221 Southern Street~दो दो दो एक सदर्न स्ट्रीट -७०० ओक स्ट्रीट~सात शून्य शून्य ओक स्ट्रीट -625 स्कूल स्ट्रीट~छह दो पाँच स्कूल स्ट्रीट +शॉप नंबर 109 9 और 10 डिवाइडिंग रोड सेक्टर 10 फरीदाबाद~शॉप नंबर एक सौ नौ नौ और दस डिवाइडिंग रोड सेक्टर दस फरीदाबाद +बूथ ७०, सेक्टर ८, चंडीगढ़~बूथ सत्तर, सेक्टर आठ, चंडीगढ़ +७०० ओक स्ट्रीट~सात सौ ओक स्ट्रीट +625 स्कूल स्ट्रीट~छह सौ पच्चीस स्कूल स्ट्रीट १४७० एस वाशिंगटन स्ट्रीट~एक चार सात शून्य एस वाशिंगटन स्ट्रीट -506 स्टेट रोड~पाँच शून्य छह स्टेट रोड -६६-४ पार्कहर्स्ट आर डी~छह छह हाइफ़न चार पार्कहर्स्ट आर डी -579 ट्रॉय-शेंक्टाडी रोड~पाँच सात नौ ट्रॉय हाइफ़न शेंक्टाडी रोड +506 स्टेट रोड~पाँच सौ छह स्टेट रोड +579 ट्रॉय-शेंक्टाडी रोड~पाँच सौ उनासी ट्रॉय हाइफ़न शेंक्टाडी रोड ७८३० - ई वेटरन्स पार्कवे, कोलंबस, जी ए ३१९०९~सात आठ तीन शून्य हाइफ़न ई वेटरन्स पार्कवे, कोलंबस, जी ए तीन एक नौ शून्य नौ -66-4, पार्कहर्स्ट रोड~छह छह हाइफ़न चार, पार्कहर्स्ट रोड -८४०/१, १०० फीट रोड, मेट्रो पिलर ५६-५७, इंदिरानगर, बैंगलोर~आठ चार शून्य बटा एक, एक शून्य शून्य फीट रोड, मेट्रो पिलर पाँच छह हाइफ़न पाँच सात, इंदिरानगर, बैंगलोर -17-18, राजलक्ष्मी नगर, 7th क्रॉस स्ट्रीट, 100 फीट बाईपास रोड, वेलाचेरी, चेन्नई~एक सात हाइफ़न एक आठ, राजलक्ष्मी नगर, सेवंथ क्रॉस स्ट्रीट, एक शून्य शून्य फीट बाईपास रोड, वेलाचेरी, चेन्नई +66-4, पार्कहर्स्ट रोड~छियासठ हाइफ़न चार, पार्कहर्स्ट रोड +८४०/१, १०० फीट रोड, मेट्रो पिलर ५६-५७, इंदिरानगर, बैंगलोर~आठ सौ चालीस बटा एक, एक सौ फीट रोड, मेट्रो पिलर छप्पन हाइफ़न सत्तावन, इंदिरानगर, बैंगलोर +17-18, राजलक्ष्मी नगर, 7th क्रॉस स्ट्रीट, 100 फीट बाईपास रोड, वेलाचेरी, चेन्नई~सत्रह हाइफ़न अठारह, राजलक्ष्मी नगर, सेवंथ क्रॉस स्ट्रीट, एक सौ फीट बाईपास रोड, वेलाचेरी, चेन्नई ४/५ न्यू म्युनिसिपल मार्केट रोड नंबर ५ और ६ सेन्टाक्रूज़ वेस्ट~चार बटा पाँच न्यू म्युनिसिपल मार्केट रोड नंबर पाँच और छह सेन्टाक्रूज़ वेस्ट -16/17 4th फ्लोर जवाहर नगर मटरू मंदिर रोड नंबर 2~एक छह बटा एक सात फ़ोर्थ फ्लोर जवाहर नगर मटरू मंदिर रोड नंबर दो -५/३०४ सिक्का कॉम्प्लेक्स विकास मार्ग एक्सटेंशन~पाँच बटा तीन शून्य चार सिक्का कॉम्प्लेक्स विकास मार्ग एक्सटेंशन -21/2 2nd फ्लोर 1st मेन रोड गांधी नगर~दो एक बटा दो सेकंड फ्लोर फ़र्स्ट मेन रोड गांधी नगर -नंबर २२/१८ ३rd फ्लोर सराय बोउ अली शू मार्केट~नंबर दो दो बटा एक आठ थर्ड फ्लोर सराय बोउ अली शू मार्केट -14/3, मथुरा रोड~एक चार बटा तीन, मथुरा रोड -यूनिट ३ १st फ्लोर नंबर ३७ सोलेमान खतर स्ट्रीट~यूनिट तीन फ़र्स्ट फ्लोर नंबर तीन सात सोलेमान खतर स्ट्रीट -1st फ्लोर नंबर 52 नॉर्थ अबूज़र स्ट्रीट खान ए अंसारी स्ट्रीट शरीयती स्ट्रीट 16617~फ़र्स्ट फ्लोर नंबर पाँच दो नॉर्थ अबूज़र स्ट्रीट खान ए अंसारी स्ट्रीट शरीयती स्ट्रीट एक छह छह एक सात -२०६ जय कॉम कॉम्प्लेक्स १st पोखरन रोड~दो शून्य छह जय कॉम कॉम्प्लेक्स फ़र्स्ट पोखरन रोड -नंबर 36 2nd फ्लोर सुपर 8 फेज 1 एकबतन टाउन तेहरान 13947~नंबर तीन छह सेकंड फ्लोर सुपर आठ फेज एक एकबतन टाउन तेहरान एक तीन नौ चार सात -२nd फ्लोर नंबर ८०८ आजादी स्ट्रीट~सेकंड फ्लोर नंबर आठ शून्य आठ आजादी स्ट्रीट -2nd फ्लोर नंबर 15 बिफ़ोर कांदि स्ट्रीट नॉर्थ सोहरावर्दी स्ट्रीट 15669~सेकंड फ्लोर नंबर एक पाँच बिफ़ोर कांदि स्ट्रीट नॉर्थ सोहरावर्दी स्ट्रीट एक पाँच छह छह नौ -यूनिट ४ नंबर २५ २nd गोलहा स्ट्रीट काशनी स्ट्रीट नूर स्क्वेर~यूनिट चार नंबर दो पाँच सेकंड गोलहा स्ट्रीट काशनी स्ट्रीट नूर स्क्वेर -ईस्ट 3rd फ्लोर नंबर 70 नेक्स्ट दो तोहीद इंस्टीट्यूट परचम स्ट्रीट~ईस्ट थर्ड फ्लोर नंबर सात शून्य नेक्स्ट दो तोहीद इंस्टीट्यूट परचम स्ट्रीट +16/17 4th फ्लोर जवाहर नगर मटरू मंदिर रोड नंबर 2~सोलह बटा सत्रह फ़ोर्थ फ्लोर जवाहर नगर मटरू मंदिर रोड नंबर दो +५/३०४ सिक्का कॉम्प्लेक्स विकास मार्ग एक्सटेंशन~पाँच बटा तीन सौ चार सिक्का कॉम्प्लेक्स विकास मार्ग एक्सटेंशन +21/2 2nd फ्लोर 1st मेन रोड गांधी नगर~इक्कीस बटा दो सेकंड फ्लोर फ़र्स्ट मेन रोड गांधी नगर +नंबर २२/१८ ३rd फ्लोर सराय बोउ अली शू मार्केट~नंबर बाईस बटा अठारह थर्ड फ्लोर सराय बोउ अली शू मार्केट +14/3, मथुरा रोड~चौदह बटा तीन, मथुरा रोड +यूनिट ३ १st फ्लोर नंबर ३७ सोलेमान खतर स्ट्रीट~यूनिट तीन फ़र्स्ट फ्लोर नंबर सैंतीस सोलेमान खतर स्ट्रीट +1st फ्लोर नंबर 52 नॉर्थ अबूज़र स्ट्रीट खान ए अंसारी स्ट्रीट शरीयती स्ट्रीट 16617~फ़र्स्ट फ्लोर नंबर बावन नॉर्थ अबूज़र स्ट्रीट खान ए अंसारी स्ट्रीट शरीयती स्ट्रीट एक छह छह एक सात +२०६ जय कॉम कॉम्प्लेक्स १st पोखरन रोड~दो सौ छह जय कॉम कॉम्प्लेक्स फ़र्स्ट पोखरन रोड +नंबर 36 2nd फ्लोर सुपर 8 फेज 1 एकबतन टाउन तेहरान 13947~नंबर छत्तीस सेकंड फ्लोर सुपर आठ फेज एक एकबतन टाउन तेहरान एक तीन नौ चार सात +२nd फ्लोर नंबर ८०८ आजादी स्ट्रीट~सेकंड फ्लोर नंबर आठ सौ आठ आजादी स्ट्रीट +2nd फ्लोर नंबर 15 बिफ़ोर कांदि स्ट्रीट नॉर्थ सोहरावर्दी स्ट्रीट 15669~सेकंड फ्लोर नंबर पंद्रह बिफ़ोर कांदि स्ट्रीट नॉर्थ सोहरावर्दी स्ट्रीट एक पाँच छह छह नौ +यूनिट ४ नंबर २५ २nd गोलहा स्ट्रीट काशनी स्ट्रीट नूर स्क्वेर~यूनिट चार नंबर पच्चीस सेकंड गोलहा स्ट्रीट काशनी स्ट्रीट नूर स्क्वेर +ईस्ट 3rd फ्लोर नंबर 70 नेक्स्ट दो तोहीद इंस्टीट्यूट परचम स्ट्रीट~ईस्ट थर्ड फ्लोर नंबर सत्तर नेक्स्ट दो तोहीद इंस्टीट्यूट परचम स्ट्रीट ३rd फ्लोर नंबर ५ हमेदन एली अपोज़िट लाले पार्क नॉर्थ कारगर स्ट्रीट~थर्ड फ्लोर नंबर पाँच हमेदन एली अपोज़िट लाले पार्क नॉर्थ कारगर स्ट्रीट 4th फ्लोर नंबर 1124 जमहोरी स्ट्रीट~फ़ोर्थ फ्लोर नंबर एक एक दो चार जमहोरी स्ट्रीट ५th फ्लोर नंबर ७/१ १३th एली शाहिद अराबली स्ट्रीट~फ़िफ्थ फ्लोर नंबर सात बटा एक थर्टींथ एली शाहिद अराबली स्ट्रीट -11, 80 फीट रोड, इंडियन ऑयल पेट्रोल पंप, कोरमंगला 6th ब्लॉक, बैंगलोर के सामने~एक एक, आठ शून्य फीट रोड, इंडियन ऑयल पेट्रोल पंप, कोरमंगला सिक्स्थ ब्लॉक, बैंगलोर के सामने -२१/११, जे ब्लॉक, ६th एवेन्यू मेन रोड, अन्ना नगर पूर्व, चेन्नई~दो एक बटा एक एक, जे ब्लॉक, सिक्स्थ एवेन्यू मेन रोड, अन्ना नगर पूर्व, चेन्नई -32A नाज़ प्लाज़ा मेरिस रोड~तीन दो ए नाज़ प्लाज़ा मेरिस रोड -२१४ बी गोविंद पूरी स्ट्रीट नंबर २~दो एक चार बी गोविंद पूरी स्ट्रीट नंबर दो -4362 16वीं एवेन्यू एसडब्ल्यू, देवदार रैपिड्स, आई ए 52404~चार तीन छह दो सोलहवीं एवेन्यू एसडब्ल्यू, देवदार रैपिड्स, आई ए बावन हज़ार चार सौ चार +11, 80 फीट रोड, इंडियन ऑयल पेट्रोल पंप, कोरमंगला 6th ब्लॉक, बैंगलोर के सामने~ग्यारह, अस्सी फीट रोड, इंडियन ऑयल पेट्रोल पंप, कोरमंगला सिक्स्थ ब्लॉक, बैंगलोर के सामने +२१/११, जे ब्लॉक, ६th एवेन्यू मेन रोड, अन्ना नगर पूर्व, चेन्नई~इक्कीस बटा ग्यारह, जे ब्लॉक, सिक्स्थ एवेन्यू मेन रोड, अन्ना नगर पूर्व, चेन्नई +32A नाज़ प्लाज़ा मेरिस रोड~बत्तीस ए नाज़ प्लाज़ा मेरिस रोड +२१४ बी गोविंद पूरी स्ट्रीट नंबर २~दो सौ चौदह बी गोविंद पूरी स्ट्रीट नंबर दो +२५१३ ५३ एवेन्यू, मुंबई, महाराष्ट्र ४००००१~दो पाँच एक तीन तिरेपन एवेन्यू, मुंबई, महाराष्ट्र चार शून्य शून्य शून्य शून्य एक अमरावती ६५५९३०~अमरावती छह पाँच पाँच नौ तीन शून्य शिमला, हिमाचल प्रदेश 593988~शिमला, हिमाचल प्रदेश पाँच नौ तीन नौ आठ आठ -२७०४४० डॉसन आर डी, अल्बानी, जीए ३१७०७~दो सात शून्य चार चार शून्य डॉसन आर डी, अल्बानी, जीए तीन एक सात शून्य सात रांची, झारखंड 736557~रांची, झारखंड सात तीन छह पाँच पाँच सात कोहिमा, नागालैंड ४४८३७७~कोहिमा, नागालैंड चार चार आठ तीन सात सात मुंबई, महाराष्ट्र 839488~मुंबई, महाराष्ट्र आठ तीन नौ चार आठ आठ diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_cardinal.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_cardinal.txt index d607992d7..050310f9f 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_cardinal.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_cardinal.txt @@ -148,3 +148,14 @@ ०७३~शून्य सात तीन 0001~शून्य शून्य शून्य एक ०००~शून्य शून्य शून्य +3,24,50,000~तीन करोड़ चौबीस लाख पचास हज़ार +२,१२,१५,०००~दो करोड़ बारह लाख पंद्रह हज़ार +32,450,000~तीन करोड़ चौबीस लाख पचास हज़ार +४,९९,९९,०००~चार करोड़ निन्यानबे लाख निन्यानबे हज़ार +5,50,00,000~पाँच करोड़ पचास लाख +32,45,000~बत्तीस लाख पैंतालीस हज़ार +५,५६,३२०~पाँच लाख छप्पन हज़ार तीन सौ बीस +1,23,456~एक लाख तेईस हज़ार चार सौ छप्पन +12,345~बारह हज़ार तीन सौ पैंतालीस +११,२२०~ग्यारह हज़ार दो सौ बीस +1,00,00,000~एक करोड़ \ No newline at end of file diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_electronic.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_electronic.txt index fd1bf459d..3265724a3 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_electronic.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_electronic.txt @@ -1,50 +1,50 @@ -gmail.com~जीमेल डॉट कॉम -yahoo.com~याहू डॉट कॉम -hotmail.com~हॉटमेल डॉट कॉम -google.com~गूगल डॉट कॉम -kumaar.org~के यू एम ए ए आर डॉट ऑर्ग -kumaar.info~के यू एम ए ए आर डॉट इन्फो -kumar@gmail.com~के यू एम ए आर एट जीमेल डॉट कॉम -robin@hotmail.com~आर ओ बी आई एन एट हॉटमेल डॉट कॉम -kapil@live.com~के ए पी आई एल एट लाइव डॉट कॉम -sneha@live.com~एस एन ई एच ए एट लाइव डॉट कॉम -mayank@google.com~एम ए वाई ए एन के एट गूगल डॉट कॉम -charu@yahoo.com~सी एच ए आर यू एट याहू डॉट कॉम -john20@yahoo.com~जे ओ एच एन दो शून्य एट याहू डॉट कॉम -vivaan62@gmail.com~वी आई वी ए ए एन छह दो एट जीमेल डॉट कॉम -viaan15@kumaar.com~वी आई ए ए एन एक पाँच एट के यू एम ए ए आर डॉट कॉम -ltaa12@gmail.com~एल टी ए ए एक दो एट जीमेल डॉट कॉम -kristen11@hotmail.com~के आर आई एस टी ई एन एक एक एट हॉटमेल डॉट कॉम -dsmith@yahoo.com~डी एस एम आई टी एच एट याहू डॉट कॉम -hgarza@gmail.com~एच जी ए आर ज़ेड ए एट जीमेल डॉट कॉम -qhill@yahoo.com~क्यू एच आई एल एल एट याहू डॉट कॉम -green-turner.org~जी आर ई ई एन हाइफ़न टी यू आर एन ई आर डॉट ऑर्ग -sharma-badami.com~एस एच ए आर एम ए हाइफ़न बी ए डी ए एम आई डॉट कॉम -osborne-gross.com~ओ एस बी ओ आर एन ई हाइफ़न जी आर ओ एस एस डॉट कॉम -lucero-stevenson.net~एल यू सी ई आर ओ हाइफ़न एस टी ई वी ई एन एस ओ एन डॉट नेट -https://google.com/~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश गूगल डॉट कॉम फॉरवर्ड स्लैश -https://github.com/~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश गिटहब डॉट कॉम फॉरवर्ड स्लैश -https://wikipedia.org/~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश विकिपीडिया डॉट ऑर्ग फॉरवर्ड स्लैश -https://amazon.com/~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश अमेज़ोन डॉट कॉम फॉरवर्ड स्लैश -www.google.com~डब्ल्यू डब्ल्यू डब्ल्यू डॉट गूगल डॉट कॉम -https://www.ndtv.com~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश डब्ल्यू डब्ल्यू डब्ल्यू डॉट एन डी टी वी डॉट कॉम -https://www.rbi.org.in/~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश डब्ल्यू डब्ल्यू डब्ल्यू डॉट आर बी आई डॉट ऑर्ग डॉट इन फॉरवर्ड स्लैश -https://www.amity.edu~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश डब्ल्यू डब्ल्यू डब्ल्यू डॉट ए एम आई टी वाई डॉट ई डी यू -https://example.com/blog/~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश ई एक्स ए एम पी एल ई डॉट कॉम फॉरवर्ड स्लैश ब्लॉग फॉरवर्ड स्लैश -https://example.com/about.html~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश ई एक्स ए एम पी एल ई डॉट कॉम फॉरवर्ड स्लैश अबाउट डॉट एच टी एम एल -https://example.com/search.php~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश ई एक्स ए एम पी एल ई डॉट कॉम फॉरवर्ड स्लैश सर्च डॉट पी एच पी -http://ati.edu~एच टी टी पी कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश ए टी आई डॉट ई डी यू -http://gcu.edu~एच टी टी पी कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश जी सी यू डॉट ई डी यू -http://pima.edu~एच टी टी पी कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश पी आई एम ए डॉट ई डी यू -bamu.nic.in/~बी ए एम यू डॉट एन आई सी डॉट इन फॉरवर्ड स्लैश -bieap.gov.in/~बी आई ई ए पी डॉट जी ओ वी डॉट इन फॉरवर्ड स्लैश -www.sharda.ac.in~डब्ल्यू डब्ल्यू डब्ल्यू डॉट एस एच ए आर डी ए डॉट ए सी डॉट इन -C:\Users\HP\Desktop\~सी कोलन बैकवर्ड स्लैश यू एस ई आर एस बैकवर्ड स्लैश एच पी बैकवर्ड स्लैश डी ई एस के टी ओ पी बैकवर्ड स्लैश -C:\Users\HP\Downloads\~सी कोलन बैकवर्ड स्लैश यू एस ई आर एस बैकवर्ड स्लैश एच पी बैकवर्ड स्लैश डी ओ डब्ल्यू एन एल ओ ए डी एस बैकवर्ड स्लैश -C:\Users\HP\Documents\Zoom~सी कोलन बैकवर्ड स्लैश यू एस ई आर एस बैकवर्ड स्लैश एच पी बैकवर्ड स्लैश डी ओ सी यू एम ई एन टी एस बैकवर्ड स्लैश ज़ेड ओ ओ एम -/home/desktop~फॉरवर्ड स्लैश होम फॉरवर्ड स्लैश डेस्कटॉप -/etc/apache~फॉरवर्ड स्लैश ई टी सी फॉरवर्ड स्लैश ए पी ए सी एच ई -/var/www~फॉरवर्ड स्लैश वी ए आर फॉरवर्ड स्लैश डब्ल्यू डब्ल्यू डब्ल्यू +gmail.com~gmail डॉट com +yahoo.com~yahoo डॉट com +hotmail.com~hotmail डॉट com +google.com~google डॉट com +kumaar.org~kumaar डॉट org +kumaar.info~kumaar डॉट info +kumar@gmail.com~kumar एट gmail डॉट com +robin@hotmail.com~robin एट hotmail डॉट com +kapil@live.com~kapil एट live डॉट com +sneha@live.com~sneha एट live डॉट com +mayank@google.com~mayank एट google डॉट com +charu@yahoo.com~charu एट yahoo डॉट com +john20@yahoo.com~john दो शून्य एट yahoo डॉट com +vivaan62@gmail.com~vivaan छह दो एट gmail डॉट com +viaan15@kumaar.com~viaan एक पाँच एट kumaar डॉट com +ltaa12@gmail.com~ltaa एक दो एट gmail डॉट com +kristen11@hotmail.com~kristen एक एक एट hotmail डॉट com +dsmith@yahoo.com~dsmith एट yahoo डॉट com +hgarza@gmail.com~hgarza एट gmail डॉट com +qhill@yahoo.com~qhill एट yahoo डॉट com +green-turner.org~green हाइफ़न turner डॉट org +sharma-badami.com~sharma हाइफ़न badami डॉट com +osborne-gross.com~osborne हाइफ़न gross डॉट com +lucero-stevenson.net~lucero हाइफ़न stevenson डॉट net +https://google.com/~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश google डॉट com फॉरवर्ड स्लैश +https://github.com/~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश github डॉट com फॉरवर्ड स्लैश +https://wikipedia.org/~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश wikipedia डॉट org फॉरवर्ड स्लैश +https://amazon.com/~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश amazon डॉट com फॉरवर्ड स्लैश +www.google.com~www डॉट google डॉट com +https://www.ndtv.com~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश www डॉट ndtv डॉट com +https://www.rbi.org.in/~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश www डॉट rbi डॉट org डॉट in फॉरवर्ड स्लैश +https://www.amity.edu~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश www डॉट amity डॉट edu +https://example.com/blog/~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश example डॉट com फॉरवर्ड स्लैश blog फॉरवर्ड स्लैश +https://example.com/about.html~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश example डॉट com फॉरवर्ड स्लैश about डॉट html +https://example.com/search.php~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश example डॉट com फॉरवर्ड स्लैश search डॉट php +http://ati.edu~http कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश ati डॉट edu +http://gcu.edu~http कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश gcu डॉट edu +http://pima.edu~http कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश pima डॉट edu +bamu.nic.in/~bamu डॉट nic डॉट in फॉरवर्ड स्लैश +bieap.gov.in/~bieap डॉट gov डॉट in फॉरवर्ड स्लैश +www.sharda.ac.in~www डॉट sharda डॉट ac डॉट in +C:\Users\HP\Desktop~C कोलन बैकवर्ड स्लैश Users बैकवर्ड स्लैश HP बैकवर्ड स्लैश Desktop +C:\Users\HP\Downloads~C कोलन बैकवर्ड स्लैश Users बैकवर्ड स्लैश HP बैकवर्ड स्लैश Downloads +C:\Users\HP\Documents\Zoom~C कोलन बैकवर्ड स्लैश Users बैकवर्ड स्लैश HP बैकवर्ड स्लैश Documents बैकवर्ड स्लैश Zoom +/home/desktop~फॉरवर्ड स्लैश home फॉरवर्ड स्लैश desktop +/etc/apache~फॉरवर्ड स्लैश etc फॉरवर्ड स्लैश apache +/var/www~फॉरवर्ड स्लैश var फॉरवर्ड स्लैश www 192.168.1.1~एक नौ दो डॉट एक छह आठ डॉट एक डॉट एक 10.0.0.1~एक शून्य डॉट शून्य डॉट शून्य डॉट एक 83.54.245.61~आठ तीन डॉट पाँच चार डॉट दो चार पाँच डॉट छह एक @@ -55,11 +55,11 @@ C:\Users\HP\Documents\Zoom~सी कोलन बैकवर्ड स्ल आईपी एड्रेस 10.0.0.1~आईपी एड्रेस एक शून्य डॉट शून्य डॉट शून्य डॉट एक ip address 192.168.1.1~ip address एक नौ दो डॉट एक छह आठ डॉट एक डॉट एक ip address 10.0.0.1~ip address एक शून्य डॉट शून्य डॉट शून्य डॉट एक -report.pdf~आर ई पी ओ आर टी डॉट पी डी एफ -photo.jpg~पी एच ओ टी ओ डॉट जे पी जी -data.csv~डेटा डॉट सी एस वी -robinson.org~आर ओ बी आई एन एस ओ एन डॉट ऑर्ग -anand@gmail.com~ए एन ए एन डी एट जीमेल डॉट कॉम +report.pdf~report डॉट pdf +photo.jpg~photo डॉट jpg +data.csv~data डॉट csv +robinson.org~robinson डॉट org +anand@gmail.com~anand एट gmail डॉट com Al₂(SO₄)₃~ए एल दो ओपन ब्रेकेट एस ओ चार क्लोज़ ब्रेकेट तीन C₂H₄~सी दो एच चार -home/desktop~होम फॉरवर्ड स्लैश डेस्कटॉप \ No newline at end of file +home/desktop~home फॉरवर्ड स्लैश desktop \ No newline at end of file diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_money.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_money.txt index 26b292a17..e5f157872 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_money.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_money.txt @@ -125,4 +125,20 @@ $1.2000~एक डॉलर बीस सेंट ₹5.00~पाँच रुपए ₹१~एक रुपया ₹२.१२३~दो दशमलव एक दो तीन रुपए -₹१.१२३४~एक दशमलव एक दो तीन चार रुपए \ No newline at end of file +₹१.१२३४~एक दशमलव एक दो तीन चार रुपए +₹3,24,50,000~तीन करोड़ चौबीस लाख पचास हज़ार रुपए +₹32,450,000~तीन करोड़ चौबीस लाख पचास हज़ार रुपए +₹5,50,00,000~पाँच करोड़ पचास लाख रुपए +₹12,54,000~बारह लाख चौवन हज़ार रुपए +₹1,00,000~एक लाख रुपए +₹2,148~दो हज़ार एक सौ अड़तालीस रुपए +₹99,999~निन्यानबे हज़ार नौ सौ निन्यानबे रुपए +₹३,२४,५०,०००~तीन करोड़ चौबीस लाख पचास हज़ार रुपए +₹३२,४५०,०००~तीन करोड़ चौबीस लाख पचास हज़ार रुपए +₹२,१२,१५,०००~दो करोड़ बारह लाख पंद्रह हज़ार रुपए +₹५,५०,००,०००~पाँच करोड़ पचास लाख रुपए +₹१२,५४,०००~बारह लाख चौवन हज़ार रुपए +₹५,५६,३२०~पाँच लाख छप्पन हज़ार तीन सौ बीस रुपए +₹१,००,०००~एक लाख रुपए +₹२,१४८~दो हज़ार एक सौ अड़तालीस रुपए +₹९९,९९९~निन्यानबे हज़ार नौ सौ निन्यानबे रुपए \ No newline at end of file diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_roman.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_roman.txt index 00f697a89..340c754ed 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_roman.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_roman.txt @@ -12,7 +12,7 @@ राष्ट्रीय राजमार्ग-IV~राष्ट्रीय राजमार्ग चार रोहिणी आर एस-I~रोहिणी आर एस एक पीएसएलवी सी-IV~पीएसएलवी सी चार -ISRO मिशन-III~आई एस आर ओ मिशन तीन +ISRO मिशन-III~ISRO मिशन तीन कक्षा XII की परीक्षा~कक्षा बारह की परीक्षा XIIवीं कक्षा की परीक्षा~बारहवीं कक्षा की परीक्षा भाग II का सारांश~भाग दो का सारांश diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_serial.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_serial.txt index 4c3880fb9..0ca21296c 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_serial.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_serial.txt @@ -18,4 +18,13 @@ 10-20-30~दस-बीस-तीस 1-800-999~एक-आठ सौ-नौ सौ निन्यानबे पृथ्वी-4~पृथ्वी-चार -ब्रह्मोस-1~ब्रह्मोस-एक \ No newline at end of file +ब्रह्मोस-1~ब्रह्मोस-एक +Q1~क्यू एक +A10~ए दस +A12~ए बारह +B-60~बी-साठ +ABC-123~ए बी सी-एक सौ तेईस +FY2024~एफ वाई दो शून्य दो चार +H2O~एच दो ओ +CO2~सी ओ दो +ABCDE1234F~ए बी सी डी ई एक दो तीन चार एफ \ No newline at end of file From e7098b3d6ecc8dae1307fab20c3c9e68b98fcef5 Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Fri, 24 Jul 2026 17:12:58 +0000 Subject: [PATCH 2/2] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- .../text_normalization/hi/taggers/cardinal.py | 4 +--- .../text_normalization/hi/taggers/measure.py | 21 ++++++------------- .../text_normalization/hi/taggers/serial.py | 12 +++-------- 3 files changed, 10 insertions(+), 27 deletions(-) diff --git a/nemo_text_processing/text_normalization/hi/taggers/cardinal.py b/nemo_text_processing/text_normalization/hi/taggers/cardinal.py index 96afdf885..d26016093 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/cardinal.py +++ b/nemo_text_processing/text_normalization/hi/taggers/cardinal.py @@ -360,9 +360,7 @@ def create_larger_number_graph(digit_graph, suffix, zeros_counts, sub_graph): + three_digits ) # International grouping: 1-3 leading digits, one or more groups of 3. - western_grouping = pynini.closure(NEMO_ALL_DIGIT, 1, 3) + pynini.closure( - delete_separator + three_digits, 1 - ) + western_grouping = pynini.closure(NEMO_ALL_DIGIT, 1, 3) + pynini.closure(delete_separator + three_digits, 1) strip_separators = (indian_grouping | western_grouping).optimize() cardinal_with_separators = pynini.compose(strip_separators, graph_without_leading_zeros).optimize() diff --git a/nemo_text_processing/text_normalization/hi/taggers/measure.py b/nemo_text_processing/text_normalization/hi/taggers/measure.py index 563c1e700..c8c8c5b54 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/measure.py +++ b/nemo_text_processing/text_normalization/hi/taggers/measure.py @@ -98,9 +98,7 @@ def get_structured_address_graph(self, cardinal: GraphFst, ordinal: GraphFst, in pincode = (num_token + pynini.closure(insert_space + num_token, 5, 5)).optimize() # Street number: 1-3 digits read as cardinal, 4+ digits read digit-by-digit - num_1to3 = ( - cardinal.digit | cardinal.zero | cardinal.teens_and_ties | cardinal.graph_hundreds - ).optimize() + num_1to3 = (cardinal.digit | cardinal.zero | cardinal.teens_and_ties | cardinal.graph_hundreds).optimize() num_4plus = (num_token + pynini.closure(insert_space + num_token, 3)).optimize() street_num = (num_1to3 | num_4plus).optimize() @@ -135,7 +133,7 @@ def get_structured_address_graph(self, cardinal: GraphFst, ordinal: GraphFst, in def get_address_graph(self, cardinal: GraphFst, ordinal: GraphFst, input_case: str): """ Address tagger that fires when address context keywords are present. - + Examples: "७०० ओक स्ट्रीट" -> "सात सौ ओक स्ट्रीट" "६६-४ पार्क रोड" -> "छियासठ हाइफ़न चार पार्क रोड" @@ -145,12 +143,7 @@ def get_address_graph(self, cardinal: GraphFst, ordinal: GraphFst, input_case: s # Strip internal weights from ordinal graph so a small outer weight suffices ordinal_graph = pynini.arcmap(ordinal.graph, map_type="rmweight").optimize() # Alphanumeric to word mappings (digits, special characters, telephone digits) - char_to_word = ( - digit - | zero - | special_characters_map - | telephone_number - ).optimize() + char_to_word = (digit | zero | special_characters_map | telephone_number).optimize() letter_to_word = capitalized_input_graph(letters_map) # Identity acceptor for keywords (Devanagari/English) to prevent unintended rewrites/transliteration address_keywords = pynini.project( @@ -183,11 +176,9 @@ def get_address_graph(self, cardinal: GraphFst, ordinal: GraphFst, input_case: s # --- Alphanumeric codes (letter+digit): letters -> Devanagari, digits via cross-class rule --- code_letter = pynini.compose(single_letter, letter_to_word).optimize() code_letters = code_letter + pynini.closure(insert_space + code_letter) - code_num_1to3 = ( - cardinal.digit | cardinal.zero | cardinal.teens_and_ties | cardinal.graph_hundreds - ).optimize() + code_num_1to3 = (cardinal.digit | cardinal.zero | cardinal.teens_and_ties | cardinal.graph_hundreds).optimize() code_num_4plus = pynini.compose( - single_digit ** 4 + pynini.closure(single_digit), cardinal.single_digits_graph + single_digit**4 + pynini.closure(single_digit), cardinal.single_digits_graph ).optimize() code_num = (code_num_1to3 | code_num_4plus).optimize() code_seg = (code_letters | code_num).optimize() @@ -211,7 +202,7 @@ def get_address_graph(self, cardinal: GraphFst, ordinal: GraphFst, input_case: s code_transliterate = pynini.arcmap(code_transliterate, map_type="rmweight").optimize() code_processor = insert_space + code_transliterate - # Pure numeric runs: 1-3 digits read as cardinal, 4+ digits read digit-by-digit + # Pure numeric runs: 1-3 digits read as cardinal, 4+ digits read digit-by-digit number_run = pynini.compose(pynini.closure(single_digit, 1), code_num).optimize() number_run_processor = pynutil.add_weight(insert_space + number_run, 0.5) diff --git a/nemo_text_processing/text_normalization/hi/taggers/serial.py b/nemo_text_processing/text_normalization/hi/taggers/serial.py index 77578ec4c..346340eb2 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/serial.py +++ b/nemo_text_processing/text_normalization/hi/taggers/serial.py @@ -67,14 +67,14 @@ def __init__( # Number-group sizing for codes: 1-3 digit groups are read as cardinals and 4+ digits are read as digit by digit digitwise_4plus = pynini.compose( - any_digit ** 4 + pynini.closure(any_digit), cardinal.single_digits_graph + any_digit**4 + pynini.closure(any_digit), cardinal.single_digits_graph ).optimize() num_graph = (limited_cardinal_graph | digitwise_4plus).optimize() symbols_graph = pynini.string_file(get_abs_path("data/serial/special_symbols.tsv")).optimize() devanagari_chars = pynini.string_file(get_abs_path("data/serial/chars.tsv")).optimize() - + letter_graph = pynini.string_file(get_abs_path("data/address/letters.tsv")) letter_graph = (letter_graph | pynini.compose(TO_LOWER, letter_graph)).optimize() latin_letters = letter_graph + pynini.closure(pynutil.insert(" ") + letter_graph) @@ -149,13 +149,7 @@ def __init__( + pynini.union(date_year_suffix, date_suffixes) ) - exclusions = ( - pure_word_slash - | pure_latin_word - | dimension_pattern - | ordinal_pattern - | date_pattern - ) + exclusions = pure_word_slash | pure_latin_word | dimension_pattern | ordinal_pattern | date_pattern accepted_inputs = pynini.difference(NEMO_SIGMA, exclusions).optimize() serial_graph = pynini.compose(accepted_inputs, serial_graph).optimize()