diff --git a/Jenkinsfile b/Jenkinsfile index a6abeb361..59de7b2a8 100644 --- a/Jenkinsfile +++ b/Jenkinsfile @@ -28,7 +28,7 @@ pipeline { MR_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/03-12-24-1' JA_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/10-17-24-1' KO_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/04-23-26-0' - HI_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/07-10-26-0' + HI_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/07-25-26-0' DEFAULT_TN_CACHE='/home/jenkins/TestData/text_norm/ci/grammars/06-08-23-0' } stages { diff --git a/nemo_text_processing/text_normalization/hi/data/address/context.tsv b/nemo_text_processing/text_normalization/hi/data/address/context.tsv index 9faadaa3b..d57bfd7d3 100644 --- a/nemo_text_processing/text_normalization/hi/data/address/context.tsv +++ b/nemo_text_processing/text_normalization/hi/data/address/context.tsv @@ -44,5 +44,4 @@ वेस्ट सामने पीछे -वीया -आर डी \ No newline at end of file +वीया \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/address/en_to_hi_mapping.tsv b/nemo_text_processing/text_normalization/hi/data/address/en_to_hi_mapping.tsv deleted file mode 100644 index 15929b547..000000000 --- a/nemo_text_processing/text_normalization/hi/data/address/en_to_hi_mapping.tsv +++ /dev/null @@ -1,2 +0,0 @@ -street स्ट्रीट -southern सदर्न \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/electronic/common_words.tsv b/nemo_text_processing/text_normalization/hi/data/electronic/common_words.tsv deleted file mode 100644 index a9f8e937d..000000000 --- a/nemo_text_processing/text_normalization/hi/data/electronic/common_words.tsv +++ /dev/null @@ -1,62 +0,0 @@ -about अबाउट -blog ब्लॉग -home होम -index इंडेक्स -login लॉगिन -register रजिस्टर -search सर्च -tags टैग्स -category केटेगरी -categories केटेगरीज़ -post पोस्ट -posts पोस्ट्स -page पेज -pages पेजेस -user यूज़र -users यूज़र्स -admin एडमिन -app ऐप -help हेल्प -terms टर्म्स -privacy प्राइवेसी -contact कॉन्टैक्ट -main मेन -explore एक्सप्लोर -wiki विकी -docs डॉक्स -download डाउनलोड -downloads डाउनलोड्स -upload अपलोड -uploads अपलोड्स -photos फ़ोटोज़ -images इमेजेज़ -music म्यूज़िक -video वीडियो -videos वीडियोज़ -desktop डेस्कटॉप -documents डॉक्युमेंट्स -tests टेस्ट्स -test टेस्ट -config कॉन्फ़िग -settings सेटिंग्स -profile प्रोफ़ाइल -account अकाउंट -web वेब -email ई मेल -mobile मोबाइल -phone फोन -phones फोन्स -online ऑनलाइन -domain डोमेन -domains डोमेन्स -data डेटा -file फ़ाइल -files फाइल्स -audio ऑडियो -software सॉफ्टवेयर -homepage होमपेज -content कंटेन्ट -default डिफ़ॉल्ट -world वर्ल्ड -list लिस्ट -license लाइसेंस \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/electronic/domain.tsv b/nemo_text_processing/text_normalization/hi/data/electronic/domain.tsv index bdb1a902d..dccb5dd90 100644 --- a/nemo_text_processing/text_normalization/hi/data/electronic/domain.tsv +++ b/nemo_text_processing/text_normalization/hi/data/electronic/domain.tsv @@ -1,24 +1,24 @@ -com कॉम -org ऑर्ग -net नेट -edu ई डी यू -gov जी ओ वी -in इन -co सी ओ -io आई ओ -ai ए आई -uk यू के -us यू एस -au ए यू -ca सी ए -ac ए सी -res आर ई एस -nic एन आई सी -ernet ई आर नेट -tv टी वी -me एम ई -tech टेक -dev डी ई वी -app ऐप -biz बिज़ -info इन्फो \ No newline at end of file +com +org +net +edu +gov +in +co +io +ai +uk +us +au +ca +ac +res +nic +ernet +tv +me +tech +dev +app +biz +info diff --git a/nemo_text_processing/text_normalization/hi/data/electronic/file_extensions.tsv b/nemo_text_processing/text_normalization/hi/data/electronic/file_extensions.tsv index e768c3838..7283febf5 100644 --- a/nemo_text_processing/text_normalization/hi/data/electronic/file_extensions.tsv +++ b/nemo_text_processing/text_normalization/hi/data/electronic/file_extensions.tsv @@ -1,34 +1,34 @@ -jpg जे पी जी -jpeg जे पी ई जी -png पी एन जी -gif जी आई एफ -pdf पी डी एफ -doc डी ओ सी -docx डी ओ सी एक्स -xls एक्स एल एस -xlsx एक्स एल एस एक्स -ppt पी पी टी -pptx पी पी टी एक्स -csv सी एस वी -txt टी एक्स टी -html एच टी एम एल -xml एक्स एम एल -json जे एस ओ एन -css सी एस एस -js जे एस -py पी वाई -java जावा -cpp सी पी पी -zip ज़िप -rar आर ए आर -tar टी ए आर -mp3 एम पी तीन -mp4 एम पी चार -avi ए वी आई -mkv एम के वी -mov एम ओ वी -wav डब्ल्यू ए वी -svg एस वी जी -apk ए पी के -exe ई एक्स ई -sql एस क्यू एल \ No newline at end of file +jpg +jpeg +png +gif +pdf +doc +docx +xls +xlsx +ppt +pptx +csv +txt +html +xml +json +css +js +py +java +cpp +zip +rar +tar +mp3 +mp4 +avi +mkv +mov +wav +svg +apk +exe +sql \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/data/electronic/protocols.tsv b/nemo_text_processing/text_normalization/hi/data/electronic/protocols.tsv index 627781003..dd897a4b6 100644 --- a/nemo_text_processing/text_normalization/hi/data/electronic/protocols.tsv +++ b/nemo_text_processing/text_normalization/hi/data/electronic/protocols.tsv @@ -1,5 +1,5 @@ -https एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश -http एच टी टी पी कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश -www डब्ल्यू डब्ल्यू डब्ल्यू डॉट -httpswww एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश डब्ल्यू डब्ल्यू डब्ल्यू डॉट -httpwww एच टी टी पी कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश डब्ल्यू डब्ल्यू डब्ल्यू डॉट \ No newline at end of file +https https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश +http http कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश +www www डॉट +httpswww https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश www डॉट +httpwww http कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश www डॉट diff --git a/nemo_text_processing/text_normalization/hi/data/electronic/server_name.tsv b/nemo_text_processing/text_normalization/hi/data/electronic/server_name.tsv deleted file mode 100644 index d0029ee5c..000000000 --- a/nemo_text_processing/text_normalization/hi/data/electronic/server_name.tsv +++ /dev/null @@ -1,25 +0,0 @@ -gmail जीमेल -yahoo याहू -hotmail हॉटमेल -outlook आउटलुक -live लाइव -google गूगल -microsoft माइक्रोसॉफ्ट -facebook फ़ेसबुक -twitter ट्विटर -instagram इंस्टाग्राम -linkedin लिंक्डइन -youtube यूट्यूब -amazon अमेज़ोन -wikipedia विकिपीडिया -github गिटहब -reddit रेडिट -netflix नेटफ्लिक्स -spotify स्पॉटिफाई -apple एप्पल -samsung सैमसंग -nvidia एनविडिया -intel इंटेल -adobe अडोब -wordpress वर्डप्रेस -blogger ब्लॉगर \ No newline at end of file diff --git a/nemo_text_processing/text_normalization/hi/taggers/cardinal.py b/nemo_text_processing/text_normalization/hi/taggers/cardinal.py index c29ccaa59..dd4611010 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/cardinal.py +++ b/nemo_text_processing/text_normalization/hi/taggers/cardinal.py @@ -18,6 +18,7 @@ from nemo_text_processing.text_normalization.hi.graph_utils import ( NEMO_ALL_DIGIT, NEMO_ALL_ZERO, + NEMO_DIGIT, GraphFst, insert_space, ) @@ -348,11 +349,46 @@ def create_larger_number_graph(digit_graph, suffix, zeros_counts, sub_graph): ) cardinal_with_leading_zeros = pynutil.add_weight(cardinal_with_leading_zeros, 0.5) + # Handle large numbers written with digit-group separators. + delete_separator = pynutil.delete(",") + two_digits = NEMO_ALL_DIGIT + NEMO_ALL_DIGIT + three_digits = NEMO_ALL_DIGIT + NEMO_ALL_DIGIT + NEMO_ALL_DIGIT + # Indian grouping: 1-2 leading digits, groups of 2, final group of 3. + indian_grouping = ( + pynini.closure(NEMO_ALL_DIGIT, 1, 2) + + pynini.closure(delete_separator + two_digits) + + delete_separator + + three_digits + ) + # International grouping: 1-3 leading digits, one or more groups of 3. + western_grouping = pynini.closure(NEMO_ALL_DIGIT, 1, 3) + pynini.closure(delete_separator + three_digits, 1) + strip_separators = (indian_grouping | western_grouping).optimize() + cardinal_with_separators = pynini.compose(strip_separators, graph_without_leading_zeros).optimize() + # Full graph including leading zeros - for standalone cardinal matching - final_graph = graph_without_leading_zeros | cardinal_with_leading_zeros + final_graph = graph_without_leading_zeros | cardinal_with_leading_zeros | cardinal_with_separators optional_minus_graph = pynini.closure(pynutil.insert("negative: ") + pynini.cross("-", "\"true\" "), 0, 1) + # --- Centralized logic for Address & Serial classes --- + # 1-3 digit groups read as cardinals, 4+ digits read digit-by-digit + limited_cardinal_graph = (self.digit | self.zero | self.teens_and_ties | self.graph_hundreds).optimize() + + any_digit = pynini.union( + NEMO_DIGIT, + pynini.project( + pynini.union( + pynini.string_file(get_abs_path("data/numbers/digit.tsv")), + pynini.string_file(get_abs_path("data/numbers/zero.tsv")), + ), + "input", + ), + ).optimize() + + digitwise_4plus = pynini.compose(any_digit**4 + pynini.closure(any_digit), self.single_digits_graph).optimize() + + self.code_num_graph = (limited_cardinal_graph | digitwise_4plus).optimize() + self.final_graph = final_graph.optimize() final_graph = optional_minus_graph + pynutil.insert("integer: \"") + self.final_graph + pynutil.insert("\"") final_graph = self.add_tokens(final_graph) diff --git a/nemo_text_processing/text_normalization/hi/taggers/electronic.py b/nemo_text_processing/text_normalization/hi/taggers/electronic.py index a309ec089..e1b93835e 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/electronic.py +++ b/nemo_text_processing/text_normalization/hi/taggers/electronic.py @@ -22,7 +22,7 @@ class ElectronicFst(GraphFst): """ Finite state transducer for classifying electronic: as URLs, email addresses, file paths, - IP addresses, domains, chemical formulas, and alphanumeric codes. + IP addresses, domains, and chemical formulas. e.g. kumar@gmail.com -> tokens { electronic { username: "kumar" domain: "gmail.com" } } e.g. https://google.com/ -> tokens { electronic { protocol: "https" domain: "google.com/" } } e.g. C:\\Users\\HP\\Desktop -> tokens { electronic { path: "C:\\Users\\HP\\Desktop" } } @@ -157,22 +157,13 @@ def __init__(self, deterministic: bool = True): unbalanced_trailing = pynini.intersect(no_open, ends_with_close) valid_chemical = pynini.difference(raw_chemical, unbalanced_trailing).optimize() - chemical_formula = pynutil.insert("domain: \"") + valid_chemical + pynutil.insert("\"") + # Recognise a chemical formula only when it uses subscript notation + chem_sigma = pynini.closure(NEMO_ALPHA | NEMO_DIGIT | subscript_digit | chemical_symbols) + contains_subscript = chem_sigma + subscript_digit + chem_sigma + valid_chemical = pynini.intersect(valid_chemical, contains_subscript).optimize() - alnum_seg = pynini.closure(NEMO_ALPHA | NEMO_DIGIT, 1) - separator = pynini.accep("-") | pynini.accep(".") - alphanumeric_pattern = alnum_seg + pynini.closure(separator + alnum_seg) - - alnum_hyp_dot_sigma = pynini.closure(NEMO_ALPHA | NEMO_DIGIT | pynini.accep("-") | pynini.accep(".")) - - contains_alpha = alnum_hyp_dot_sigma + NEMO_ALPHA + alnum_hyp_dot_sigma - contains_digit = alnum_hyp_dot_sigma + NEMO_DIGIT + alnum_hyp_dot_sigma - - alphanumeric_code_fst = pynini.intersect( - pynini.intersect(alphanumeric_pattern, contains_alpha), contains_digit - ).optimize() - - alphanumeric_code = pynutil.insert("domain: \"") + alphanumeric_code_fst + pynutil.insert("\"") + # Chemical formulas carry a dedicated tag so the verbalizer can spell element + chemical_formula = pynutil.insert("fragment_id: \"") + valid_chemical + pynutil.insert("\"") graph = ( pynutil.add_weight(url_graph, 1.0) @@ -184,7 +175,6 @@ def __init__(self, deterministic: bool = True): | pynutil.add_weight(combined_domain, 1.1) | pynutil.add_weight(file_with_extension, 1.1) | pynutil.add_weight(chemical_formula, 1.2) - | pynutil.add_weight(alphanumeric_code, 1.2) ) self.graph = graph.optimize() diff --git a/nemo_text_processing/text_normalization/hi/taggers/measure.py b/nemo_text_processing/text_normalization/hi/taggers/measure.py index e18111696..8e0d0d3be 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/measure.py +++ b/nemo_text_processing/text_normalization/hi/taggers/measure.py @@ -28,12 +28,12 @@ HI_SADHE, HI_SAVVA, HYPHEN, - INPUT_LOWER_CASED, LOWERCASE_X, NEMO_CHAR, NEMO_DIGIT, NEMO_HI_DIGIT, NEMO_NOT_SPACE, + NEMO_SIGMA, NEMO_SPACE, NEMO_WHITE_SPACE, ONE_POINT_FIVE, @@ -50,11 +50,15 @@ from nemo_text_processing.text_normalization.hi.utils import get_abs_path digit = pynini.string_file(get_abs_path("data/numbers/digit.tsv")) -# Load both Hindi (Devanagari) and English (Arabic) number mappings -teens_ties_hi = pynini.string_file(get_abs_path("data/numbers/teens_and_ties.tsv")) -teens_ties_en = pynini.string_file(get_abs_path("data/numbers/teens_and_ties_en.tsv")) -teens_ties = pynini.union(teens_ties_hi, teens_ties_en) -teens_and_ties = pynutil.add_weight(teens_ties, -0.1) + +# Shared Address Maps +zero = pynini.string_file(get_abs_path("data/numbers/zero.tsv")) +telephone_number = pynini.string_file(get_abs_path("data/telephone/number.tsv")) +states_map = pynini.string_file(get_abs_path("data/address/states.tsv")) +cities_map = pynini.string_file(get_abs_path("data/address/cities.tsv")) +special_characters_map = pynini.string_file(get_abs_path("data/address/special_characters.tsv")) +letters_map = pynini.string_file(get_abs_path("data/address/letters.tsv")) +context_map = pynini.string_file(get_abs_path("data/address/context.tsv")) class MeasureFst(GraphFst): @@ -71,32 +75,27 @@ class MeasureFst(GraphFst): for False multiple transduction are generated (used for audio-based normalization) """ - def get_structured_address_graph(self, ordinal: GraphFst, input_case: str): + def get_structured_address_graph(self, cardinal: GraphFst, ordinal: GraphFst, input_case: str): """ Minimal address tagger for state/city + pincode patterns only. - Highly optimized for performance. Examples: "मुंबई ८८४४०४" -> "मुंबई आठ आठ चार चार शून्य चार" "गोवा १२३४५६" -> "गोवा एक दो तीन चार पाँच छह" + "100 फीट रोड, चेन्नई" -> "एक सौ फीट रोड, चेन्नई" """ # State/city keywords - states = pynini.string_file(get_abs_path("data/address/states.tsv")) - cities = pynini.string_file(get_abs_path("data/address/cities.tsv")) - state_city_names = pynini.union(states, cities).optimize() - - # Digit mappings - num_token = ( - digit - | pynini.string_file(get_abs_path("data/numbers/zero.tsv")) - | pynini.string_file(get_abs_path("data/telephone/number.tsv")) - ).optimize() - # Pincode (6 digits) + state_city_names = pynini.union(states_map, cities_map).optimize() + + # Digit mappings (shared maps loaded once at module level) + num_token = (digit | zero | telephone_number).optimize() + + # Pincode (6 digits) -> always digit-by-digit (length >= 4) pincode = (num_token + pynini.closure(insert_space + num_token, 5, 5)).optimize() - # Street number (1-4 digits) - street_num = (num_token + pynini.closure(insert_space + num_token, 0, 3)).optimize() + # Street number: Use centralized digit-by-digit logic from cardinal + street_num = cardinal.code_num_graph # Text: words with trailing separator (comma? + space) any_digit = pynini.union(NEMO_HI_DIGIT, NEMO_DIGIT).optimize() @@ -107,7 +106,9 @@ def get_structured_address_graph(self, ordinal: GraphFst, input_case: str): # Separator: optional comma followed by mandatory space sep = pynini.closure(pynini.accep(COMMA), 0, 1) + pynini.accep(NEMO_SPACE) word_with_sep = word + sep - text = pynini.closure(word_with_sep, 0, 5).optimize() + # Consume inline address numbers using the same 1-3 (cardinal) / 4+ (digit-by-digit) rule + num_with_sep = street_num + sep + text = pynini.closure(pynini.union(word_with_sep, num_with_sep), 0, 5).optimize() # Pattern: [street_num + sep]? text state/city [space pincode] pattern = ( @@ -124,34 +125,26 @@ def get_structured_address_graph(self, ordinal: GraphFst, input_case: str): ) return pynutil.add_weight(graph, 1.0).optimize() - def get_address_graph(self, ordinal: GraphFst, input_case: str): + def get_address_graph(self, cardinal: GraphFst, ordinal: GraphFst, serial: GraphFst, input_case: str): """ - Address tagger that converts digits/hyphens/slashes character-by-character - when address context keywords are present. - English words and ordinals are converted to Hindi transliterations. + Address tagger that fires when address context keywords are present. Examples: - "७०० ओक स्ट्रीट" -> "सात शून्य शून्य ओक स्ट्रीट" - "६६-४ पार्क रोड" -> "छह छह हाइफ़न चार पार्क रोड" + "७०० ओक स्ट्रीट" -> "सात सौ ओक स्ट्रीट" + "६६-४ पार्क रोड" -> "छियासठ हाइफ़न चार पार्क रोड" + "593988" (6-digit pincode) -> "पाँच नौ तीन नौ आठ आठ" + "32A नाज़ प्लाज़ा" -> "बत्तीस ए नाज़ प्लाज़ा" """ - ordinal_graph = ordinal.graph + # Strip internal weights from ordinal graph so a small outer weight suffices + ordinal_graph = pynini.arcmap(ordinal.graph, map_type="rmweight").optimize() # Alphanumeric to word mappings (digits, special characters, telephone digits) - char_to_word = ( - digit - | pynini.string_file(get_abs_path("data/numbers/zero.tsv")) - | pynini.string_file(get_abs_path("data/address/special_characters.tsv")) - | pynini.string_file(get_abs_path("data/telephone/number.tsv")) + char_to_word = (digit | zero | special_characters_map | telephone_number).optimize() + letter_to_word = capitalized_input_graph(letters_map) + # Identity acceptor for keywords (Devanagari/English) to prevent unintended rewrites/transliteration + address_keywords = pynini.project( + capitalized_input_graph(context_map), + "input", ).optimize() - letter_to_word = pynini.string_file(get_abs_path("data/address/letters.tsv")) - letter_to_word = capitalized_input_graph(letter_to_word) - address_keywords_hi = pynini.string_file(get_abs_path("data/address/context.tsv")) - - # English address keywords with Hindi translation (case-insensitive) - en_to_hi_map = pynini.string_file(get_abs_path("data/address/en_to_hi_mapping.tsv")) - if input_case != INPUT_LOWER_CASED: - en_to_hi_map = capitalized_input_graph(en_to_hi_map) - address_keywords_en = pynini.project(en_to_hi_map, "input") - address_keywords = pynini.union(address_keywords_hi, address_keywords_en) # Alphanumeric processing: treat digits, letters, and -/ as convertible tokens single_digit = pynini.union(NEMO_DIGIT, NEMO_HI_DIGIT).optimize() @@ -162,24 +155,34 @@ def get_address_graph(self, ordinal: GraphFst, input_case: str): NEMO_CHAR, pynini.union(NEMO_WHITE_SPACE, convertible_char, pynini.accep(COMMA)) ).optimize() - # Token processors with weights: prefer ordinals and known English→Hindi words - # Delete space before comma to avoid Sparrowhawk "sil" issue - comma_processor = pynutil.add_weight(delete_space + pynini.accep(COMMA), 0.0) - ordinal_processor = pynutil.add_weight(insert_space + ordinal_graph, -5.0) - english_word_processor = pynutil.add_weight(insert_space + en_to_hi_map, -3.0) - letter_processor = pynutil.add_weight(insert_space + pynini.compose(single_letter, letter_to_word), 0.5) - digit_char_processor = pynutil.add_weight(insert_space + pynini.compose(convertible_char, char_to_word), 0.0) - other_word_processor = pynutil.add_weight(insert_space + pynini.closure(non_space_char, 1), 0.1) + comma_processor = delete_space + pynini.accep(COMMA) + ordinal_processor = insert_space + ordinal_graph + latin_word = single_letter + pynini.closure(single_letter, 1) + english_word_processor = insert_space + latin_word + letter_processor = insert_space + pynini.compose(single_letter, letter_to_word) + special_char_processor = insert_space + pynini.compose(special_chars, char_to_word) + other_word_processor = insert_space + pynini.closure(non_space_char, 1) + + code_processor = insert_space + serial.mixed_alphanum_graph + + # Pure numeric runs: 1-3 digits read as cardinal, 4+ digits read digit-by-digit + number_run = pynini.compose(pynini.closure(single_digit, 1), cardinal.code_num_graph).optimize() + + # A positive weight of 0.5 penalizes multiple uses of this arc. This forces the FST to consume all contiguous digits as ONE run (cost 0.5) instead of splitting "625" into "6" and "25" (cost 1.0). + number_run_processor = insert_space + number_run token_processor = ( - ordinal_processor - | english_word_processor - | letter_processor - | digit_char_processor - | pynini.accep(NEMO_SPACE) + pynini.accep(NEMO_SPACE) | comma_processor - | other_word_processor + | pynutil.add_weight(special_char_processor, 0.0) + | code_processor + | pynutil.add_weight(ordinal_processor, -0.5) + | pynutil.add_weight(number_run_processor, 0.5) + | pynutil.add_weight(letter_processor, 0.5) + | pynutil.add_weight(english_word_processor, 0.1) + | pynutil.add_weight(other_word_processor, 0.1) ).optimize() + full_string_processor = pynini.closure(token_processor, 1).optimize() # Window-based context matching around address keywords for robust detection @@ -201,9 +204,9 @@ def get_address_graph(self, ordinal: GraphFst, input_case: str): + address_graph + pynutil.insert('" } preserve_order: true') ) - return pynutil.add_weight(graph, 1.05).optimize() + return graph.optimize() - def __init__(self, cardinal: GraphFst, decimal: GraphFst, ordinal: GraphFst, input_case: str): + def __init__(self, cardinal: GraphFst, decimal: GraphFst, ordinal: GraphFst, serial: GraphFst, input_case: str): super().__init__(name="measure", kind="classify") cardinal_graph = ( @@ -451,8 +454,8 @@ def __init__(self, cardinal: GraphFst, decimal: GraphFst, ordinal: GraphFst, inp + pynutil.insert("\"") ) - address_graph = self.get_address_graph(ordinal, input_case) - structured_address_graph = self.get_structured_address_graph(ordinal, input_case) + address_graph = self.get_address_graph(cardinal, ordinal, serial, input_case) + structured_address_graph = self.get_structured_address_graph(cardinal, ordinal, input_case) graph = ( pynutil.add_weight(graph_decimal, 0.1) diff --git a/nemo_text_processing/text_normalization/hi/taggers/serial.py b/nemo_text_processing/text_normalization/hi/taggers/serial.py index d7433e583..3f2244dd8 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/serial.py +++ b/nemo_text_processing/text_normalization/hi/taggers/serial.py @@ -20,6 +20,7 @@ NEMO_DIGIT, NEMO_NOT_SPACE, NEMO_SIGMA, + TO_LOWER, GraphFst, convert_space, ) @@ -38,9 +39,9 @@ class SerialFst(GraphFst): e.g. 2^2 -> tokens { name: "दो स्क्वेर्ड" } e.g. 2^4 -> tokens { name: "दो टु द पावर चार" } e.g. 1-800-555 -> tokens { name: "एक-आठ सौ-पाँच सौ पचपन" } - - Note: Pure Latin-alpha + digit patterns (A12, B-60) are intentionally - excluded here so they fall through to the electronic classifier. + e.g. B-60 -> tokens { name: "बी-साठ" } + e.g. A12 -> tokens { name: "ए बारह" } + e.g. FY2024 -> tokens { name: "एफ वाई दो शून्य दो चार" } """ def __init__( @@ -60,16 +61,15 @@ def __init__( any_digit = pynini.union(NEMO_DIGIT, devanagari_digits).optimize() - limited_cardinal_graph = ( - cardinal.digit | cardinal.zero | cardinal.teens_and_ties | cardinal.graph_hundreds - ).optimize() - num_graph = limited_cardinal_graph + # Fetch centralized 1-3 vs 4+ digit logic from cardinal + num_graph = cardinal.code_num_graph symbols_graph = pynini.string_file(get_abs_path("data/serial/special_symbols.tsv")).optimize() devanagari_chars = pynini.string_file(get_abs_path("data/serial/chars.tsv")).optimize() letter_graph = pynini.string_file(get_abs_path("data/address/letters.tsv")) + letter_graph = (letter_graph | pynini.compose(TO_LOWER, letter_graph)).optimize() latin_letters = letter_graph + pynini.closure(pynutil.insert(" ") + letter_graph) latin_letters = latin_letters.optimize() @@ -94,6 +94,16 @@ def __init__( glued_serial = pynini.compose(space_inserter, serial_core).optimize() serial_graph = pynini.union(serial_graph, glued_serial).optimize() + # Reusable mixed alphanumeric-code graph + code_join_char = pynini.union(all_alphas, any_digit, pynini.accep("-"), pynini.accep("/")) + has_letter = pynini.closure(code_join_char) + all_alphas + pynini.closure(code_join_char) + has_digit = pynini.closure(code_join_char) + any_digit + pynini.closure(code_join_char) + mixed_code_only = pynini.intersect( + pynini.intersect(pynini.closure(code_join_char, 1), has_letter), has_digit + ).optimize() + + self.mixed_alphanum_graph = pynini.compose(mixed_code_only, serial_graph).optimize() + power_special = pynutil.add_weight( pynini.string_file(get_abs_path("data/serial/power_special.tsv")), -1.0 ).optimize() @@ -110,15 +120,14 @@ def __init__( pure_word_slash = pynini.closure(NEMO_ALPHA, 1) + pynini.accep("/") + pynini.closure(NEMO_ALPHA, 1) + letter_join_char = NEMO_ALPHA | pynini.accep("-") | pynini.accep("/") + contains_latin_letter = pynini.closure(letter_join_char) + NEMO_ALPHA + pynini.closure(letter_join_char) + pure_latin_word = pynini.intersect(pynini.closure(letter_join_char, 1), contains_latin_letter).optimize() + dimension_pattern = ( pynini.closure(any_digit, 1) + (pynini.accep("x") | pynini.accep("X")) + pynini.closure(any_digit, 1) ) - _opt_delim = pynini.closure(pynini.accep("-") | pynini.accep(" "), 0, 1) - latin_alphanum = (pynini.closure(NEMO_ALPHA, 1) + _opt_delim + pynini.closure(any_digit, 1)) | ( - pynini.closure(any_digit, 1) + _opt_delim + pynini.closure(NEMO_ALPHA, 1) - ) - ordinal_suffixes = pynini.project( pynini.union( pynini.string_file(get_abs_path("data/ordinal/suffixes.tsv")), @@ -143,7 +152,7 @@ def __init__( + pynini.union(date_year_suffix, date_suffixes) ) - exclusions = pure_word_slash | dimension_pattern | latin_alphanum | ordinal_pattern | date_pattern + exclusions = pure_word_slash | pure_latin_word | dimension_pattern | ordinal_pattern | date_pattern accepted_inputs = pynini.difference(NEMO_SIGMA, exclusions).optimize() serial_graph = pynini.compose(accepted_inputs, serial_graph).optimize() diff --git a/nemo_text_processing/text_normalization/hi/taggers/tokenize_and_classify.py b/nemo_text_processing/text_normalization/hi/taggers/tokenize_and_classify.py index 0e8d7fc73..04124635e 100644 --- a/nemo_text_processing/text_normalization/hi/taggers/tokenize_and_classify.py +++ b/nemo_text_processing/text_normalization/hi/taggers/tokenize_and_classify.py @@ -100,7 +100,12 @@ def __init__( ordinal = OrdinalFst(cardinal=cardinal, deterministic=deterministic) ordinal_graph = ordinal.fst - measure = MeasureFst(cardinal=cardinal, decimal=decimal, ordinal=ordinal, input_case=input_case) + serial = SerialFst(cardinal=cardinal, deterministic=deterministic) + serial_graph = serial.fst + + measure = MeasureFst( + cardinal=cardinal, decimal=decimal, ordinal=ordinal, serial=serial, input_case=input_case + ) measure_graph = measure.fst money = MoneyFst(cardinal=cardinal) @@ -125,9 +130,6 @@ def __init__( electronic = ElectronicFst(deterministic=deterministic) electronic_graph = electronic.fst - serial = SerialFst(cardinal=cardinal, deterministic=deterministic) - serial_graph = serial.fst - classify = ( pynutil.add_weight(whitelist_graph, 1.01) | pynutil.add_weight(cardinal_graph, 1.1) diff --git a/nemo_text_processing/text_normalization/hi/verbalizers/electronic.py b/nemo_text_processing/text_normalization/hi/verbalizers/electronic.py index 398c79fef..21265c788 100644 --- a/nemo_text_processing/text_normalization/hi/verbalizers/electronic.py +++ b/nemo_text_processing/text_normalization/hi/verbalizers/electronic.py @@ -16,6 +16,7 @@ from pynini.lib import pynutil from nemo_text_processing.text_normalization.hi.graph_utils import ( + NEMO_ALPHA, GraphFst, capitalized_input_graph, delete_space, @@ -27,13 +28,15 @@ class ElectronicFst(GraphFst): """ Finite state transducer for verbalizing electronic addresses. - Uses a phonetic-first approach with letter-by-letter fallback. + English words and letters are kept verbatim (Latin script); only digits and + symbols are read out in Hindi. Examples: - electronic { username: "kumar" domain: "gmail.com" } -> "के यू एम ए आर एट जीमेल डॉट कॉम" - electronic { protocol: "https" domain: "google.com/" } -> "एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश गूगल डॉट कॉम फॉरवर्ड स्लैश" - electronic { path: "C:\\Users\\HP\\Desktop" } -> "सी कोलन बैकवर्ड स्लैश यूज़र्स बैकवर्ड स्लैश एच पी बैकवर्ड स्लैश डेस्कटॉप" + electronic { username: "kumar" domain: "gmail.com" } -> "kumar एट gmail डॉट com" + electronic { protocol: "https" domain: "google.com/" } -> "https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश google डॉट com फॉरवर्ड स्लैश" + electronic { path: "C:\\Users\\HP\\Desktop" } -> "C कोलन बैकवर्ड स्लैश Users बैकवर्ड स्लैश HP बैकवर्ड स्लैश Desktop" electronic { domain: "192.168.1.1" } -> "एक नौ दो डॉट एक छह आठ डॉट एक डॉट एक" + electronic { chem: "C₂H₄" } -> "सी दो एच चार" Args: deterministic: if True will provide a single transduction option, @@ -44,11 +47,6 @@ def __init__(self, deterministic: bool = True): super().__init__(name="electronic", kind="verbalize", deterministic=deterministic) symbols_graph = pynini.string_file(get_abs_path("data/electronic/symbols.tsv")).optimize() - domain_graph = pynini.string_file(get_abs_path("data/electronic/domain.tsv")).optimize() - server_name_graph = pynini.string_file(get_abs_path("data/electronic/server_name.tsv")).optimize() - common_words_graph = pynini.string_file(get_abs_path("data/electronic/common_words.tsv")).optimize() - latin_to_hindi_graph = pynini.string_file(get_abs_path("data/address/letters.tsv")) - latin_to_hindi_graph = capitalized_input_graph(latin_to_hindi_graph).optimize() ascii_digit_graph = pynini.string_file(get_abs_path("data/telephone/number.tsv")).optimize() hindi_digit_graph = pynini.string_file(get_abs_path("data/numbers/digit.tsv")).optimize() @@ -58,25 +56,30 @@ def __init__(self, deterministic: bool = True): protocol_graph = pynini.string_file(get_abs_path("data/electronic/protocols.tsv")).optimize() - single_letter = latin_to_hindi_graph + insert_space single_digit = digit_verbalization + insert_space single_symbol = symbols_graph + insert_space single_non_alpha = pynutil.add_weight(single_symbol, 1.0) | pynutil.add_weight(single_digit, 1.0) - def make_alpha_run_verbalizer(tsv_graphs): - phonetic = pynini.union(*[pynutil.add_weight(g + insert_space, w) for g, w in tsv_graphs]) - literal = pynutil.add_weight(pynini.closure(single_letter, 1), 1.1) - return phonetic | literal + # A run of Latin letters is preserved verbatim; digits and symbols verbalize in Hindi. + alpha_run = pynini.closure(NEMO_ALPHA, 1) + insert_space - def make_content(alpha_run_verb, non_alpha_sep=None): + # Chemical formulas are spelled out letter-by-letter (element symbols are + # abbreviations, not words), while digits and symbols verbalize in Hindi. + latin_to_hindi_graph = capitalized_input_graph( + pynini.string_file(get_abs_path("data/address/letters.tsv")) + ).optimize() + chem_char = (latin_to_hindi_graph + insert_space) | single_digit | single_symbol + chem_content = pynini.closure(chem_char, 1) + + def make_content(non_alpha_sep=None): if non_alpha_sep is None: non_alpha_sep = single_non_alpha mandatory_sep = pynini.closure(non_alpha_sep, 1) return ( pynini.closure(non_alpha_sep, 0) - + pynini.closure(alpha_run_verb + mandatory_sep, 0) - + pynini.closure(alpha_run_verb, 0, 1) + + pynini.closure(alpha_run + mandatory_sep, 0) + + pynini.closure(alpha_run, 0, 1) + pynini.closure(non_alpha_sep, 0) ) @@ -84,48 +87,17 @@ def make_content(alpha_run_verb, non_alpha_sep=None): delete_domain_tag = pynutil.delete("domain: \"") delete_protocol_tag = pynutil.delete("protocol: \"") delete_path_tag = pynutil.delete("path: \"") + delete_fragment_id_tag = pynutil.delete("fragment_id: \"") delete_quote = pynutil.delete("\"") - username_alpha_run = make_alpha_run_verbalizer( - [ - (server_name_graph, 0.85), - (domain_graph, 0.87), - (common_words_graph, 0.90), - ] - ) - username_content = make_content(username_alpha_run) - username_graph = delete_username_tag + username_content + delete_quote + delete_space + pynutil.insert("एट ") - - domain_alpha_run = make_alpha_run_verbalizer( - [ - (server_name_graph, 0.85), - (domain_graph, 0.87), - (common_words_graph, 0.90), - ] - ) - - domain_alpha_run = make_alpha_run_verbalizer( - [ - (server_name_graph, 0.85), - (domain_graph, 0.87), - (common_words_graph, 0.90), - ] - ) - - domain_content = pynutil.add_weight(make_content(domain_alpha_run), 1.0) - - domain_only_graph = delete_domain_tag + domain_content + delete_quote + general_content = make_content() + username_graph = delete_username_tag + general_content + delete_quote + delete_space + pynutil.insert("एट ") + domain_only_graph = delete_domain_tag + general_content + delete_quote protocol_only_graph = delete_protocol_tag + protocol_graph + insert_space + delete_quote + delete_space + path_graph = delete_path_tag + general_content + delete_quote - path_alpha_run = make_alpha_run_verbalizer( - [ - (domain_graph, 0.87), - (common_words_graph, 0.90), - ] - ) - path_content = make_content(path_alpha_run) - path_graph = delete_path_tag + path_content + delete_quote + chem_graph = delete_fragment_id_tag + chem_content + delete_quote ip_char = single_symbol | single_digit ip_content = pynini.closure(ip_char, 1) @@ -140,6 +112,7 @@ def make_content(alpha_run_verb, non_alpha_sep=None): | pynutil.add_weight(path_graph, 1.02) | pynutil.add_weight(ip_graph, 1.03) | pynutil.add_weight(domain_only_graph, 1.04) + | pynutil.add_weight(chem_graph, 1.04) ) delete_tokens = self.delete_tokens(graph) diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_address.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_address.txt index 9989fa75c..d0554ce30 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_address.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_address.txt @@ -1,47 +1,44 @@ -700 ओक स्ट्रीट~सात शून्य शून्य ओक स्ट्रीट -११ जंगल रोड~एक एक जंगल रोड -301 पार्क एवेन्यू~तीन शून्य एक पार्क एवेन्यू -गली नंबर १७ जीएकगढ़~गली नंबर एक सात जीएकगढ़ -अदनान अपार्टमेंट फ्लैट नंबर 55~अदनान अपार्टमेंट फ्लैट नंबर पाँच पाँच +700 ओक स्ट्रीट~सात सौ ओक स्ट्रीट +११ जंगल रोड~ग्यारह जंगल रोड +301 पार्क एवेन्यू~तीन सौ एक पार्क एवेन्यू +गली नंबर १७ जीएकगढ़~गली नंबर सत्रह जीएकगढ़ +अदनान अपार्टमेंट फ्लैट नंबर 55~अदनान अपार्टमेंट फ्लैट नंबर पचपन प्लॉट नंबर ८ बालाजी मार्केट~प्लॉट नंबर आठ बालाजी मार्केट -शॉप नंबर 109 9 और 10 डिवाइडिंग रोड सेक्टर 10 फरीदाबाद~शॉप नंबर एक शून्य नौ नौ और एक शून्य डिवाइडिंग रोड सेक्टर एक शून्य फरीदाबाद -बूथ ७०, सेक्टर ८, चंडीगढ़~बूथ सात शून्य, सेक्टर आठ, चंडीगढ़ -2221 Southern Street~दो दो दो एक सदर्न स्ट्रीट -७०० ओक स्ट्रीट~सात शून्य शून्य ओक स्ट्रीट -625 स्कूल स्ट्रीट~छह दो पाँच स्कूल स्ट्रीट +शॉप नंबर 109 9 और 10 डिवाइडिंग रोड सेक्टर 10 फरीदाबाद~शॉप नंबर एक सौ नौ नौ और दस डिवाइडिंग रोड सेक्टर दस फरीदाबाद +बूथ ७०, सेक्टर ८, चंडीगढ़~बूथ सत्तर, सेक्टर आठ, चंडीगढ़ +७०० ओक स्ट्रीट~सात सौ ओक स्ट्रीट +625 स्कूल स्ट्रीट~छह सौ पच्चीस स्कूल स्ट्रीट १४७० एस वाशिंगटन स्ट्रीट~एक चार सात शून्य एस वाशिंगटन स्ट्रीट -506 स्टेट रोड~पाँच शून्य छह स्टेट रोड -६६-४ पार्कहर्स्ट आर डी~छह छह हाइफ़न चार पार्कहर्स्ट आर डी -579 ट्रॉय-शेंक्टाडी रोड~पाँच सात नौ ट्रॉय हाइफ़न शेंक्टाडी रोड +506 स्टेट रोड~पाँच सौ छह स्टेट रोड +579 ट्रॉय-शेंक्टाडी रोड~पाँच सौ उनासी ट्रॉय हाइफ़न शेंक्टाडी रोड ७८३० - ई वेटरन्स पार्कवे, कोलंबस, जी ए ३१९०९~सात आठ तीन शून्य हाइफ़न ई वेटरन्स पार्कवे, कोलंबस, जी ए तीन एक नौ शून्य नौ -66-4, पार्कहर्स्ट रोड~छह छह हाइफ़न चार, पार्कहर्स्ट रोड -८४०/१, १०० फीट रोड, मेट्रो पिलर ५६-५७, इंदिरानगर, बैंगलोर~आठ चार शून्य बटा एक, एक शून्य शून्य फीट रोड, मेट्रो पिलर पाँच छह हाइफ़न पाँच सात, इंदिरानगर, बैंगलोर -17-18, राजलक्ष्मी नगर, 7th क्रॉस स्ट्रीट, 100 फीट बाईपास रोड, वेलाचेरी, चेन्नई~एक सात हाइफ़न एक आठ, राजलक्ष्मी नगर, सेवंथ क्रॉस स्ट्रीट, एक शून्य शून्य फीट बाईपास रोड, वेलाचेरी, चेन्नई +66-4, पार्कहर्स्ट रोड~छियासठ हाइफ़न चार, पार्कहर्स्ट रोड +८४०/१, १०० फीट रोड, मेट्रो पिलर ५६-५७, इंदिरानगर, बैंगलोर~आठ सौ चालीस बटा एक, एक सौ फीट रोड, मेट्रो पिलर छप्पन हाइफ़न सत्तावन, इंदिरानगर, बैंगलोर +17-18, राजलक्ष्मी नगर, 7th क्रॉस स्ट्रीट, 100 फीट बाईपास रोड, वेलाचेरी, चेन्नई~सत्रह हाइफ़न अठारह, राजलक्ष्मी नगर, सेवंथ क्रॉस स्ट्रीट, एक सौ फीट बाईपास रोड, वेलाचेरी, चेन्नई ४/५ न्यू म्युनिसिपल मार्केट रोड नंबर ५ और ६ सेन्टाक्रूज़ वेस्ट~चार बटा पाँच न्यू म्युनिसिपल मार्केट रोड नंबर पाँच और छह सेन्टाक्रूज़ वेस्ट -16/17 4th फ्लोर जवाहर नगर मटरू मंदिर रोड नंबर 2~एक छह बटा एक सात फ़ोर्थ फ्लोर जवाहर नगर मटरू मंदिर रोड नंबर दो -५/३०४ सिक्का कॉम्प्लेक्स विकास मार्ग एक्सटेंशन~पाँच बटा तीन शून्य चार सिक्का कॉम्प्लेक्स विकास मार्ग एक्सटेंशन -21/2 2nd फ्लोर 1st मेन रोड गांधी नगर~दो एक बटा दो सेकंड फ्लोर फ़र्स्ट मेन रोड गांधी नगर -नंबर २२/१८ ३rd फ्लोर सराय बोउ अली शू मार्केट~नंबर दो दो बटा एक आठ थर्ड फ्लोर सराय बोउ अली शू मार्केट -14/3, मथुरा रोड~एक चार बटा तीन, मथुरा रोड -यूनिट ३ १st फ्लोर नंबर ३७ सोलेमान खतर स्ट्रीट~यूनिट तीन फ़र्स्ट फ्लोर नंबर तीन सात सोलेमान खतर स्ट्रीट -1st फ्लोर नंबर 52 नॉर्थ अबूज़र स्ट्रीट खान ए अंसारी स्ट्रीट शरीयती स्ट्रीट 16617~फ़र्स्ट फ्लोर नंबर पाँच दो नॉर्थ अबूज़र स्ट्रीट खान ए अंसारी स्ट्रीट शरीयती स्ट्रीट एक छह छह एक सात -२०६ जय कॉम कॉम्प्लेक्स १st पोखरन रोड~दो शून्य छह जय कॉम कॉम्प्लेक्स फ़र्स्ट पोखरन रोड -नंबर 36 2nd फ्लोर सुपर 8 फेज 1 एकबतन टाउन तेहरान 13947~नंबर तीन छह सेकंड फ्लोर सुपर आठ फेज एक एकबतन टाउन तेहरान एक तीन नौ चार सात -२nd फ्लोर नंबर ८०८ आजादी स्ट्रीट~सेकंड फ्लोर नंबर आठ शून्य आठ आजादी स्ट्रीट -2nd फ्लोर नंबर 15 बिफ़ोर कांदि स्ट्रीट नॉर्थ सोहरावर्दी स्ट्रीट 15669~सेकंड फ्लोर नंबर एक पाँच बिफ़ोर कांदि स्ट्रीट नॉर्थ सोहरावर्दी स्ट्रीट एक पाँच छह छह नौ -यूनिट ४ नंबर २५ २nd गोलहा स्ट्रीट काशनी स्ट्रीट नूर स्क्वेर~यूनिट चार नंबर दो पाँच सेकंड गोलहा स्ट्रीट काशनी स्ट्रीट नूर स्क्वेर -ईस्ट 3rd फ्लोर नंबर 70 नेक्स्ट दो तोहीद इंस्टीट्यूट परचम स्ट्रीट~ईस्ट थर्ड फ्लोर नंबर सात शून्य नेक्स्ट दो तोहीद इंस्टीट्यूट परचम स्ट्रीट +16/17 4th फ्लोर जवाहर नगर मटरू मंदिर रोड नंबर 2~सोलह बटा सत्रह फ़ोर्थ फ्लोर जवाहर नगर मटरू मंदिर रोड नंबर दो +५/३०४ सिक्का कॉम्प्लेक्स विकास मार्ग एक्सटेंशन~पाँच बटा तीन सौ चार सिक्का कॉम्प्लेक्स विकास मार्ग एक्सटेंशन +21/2 2nd फ्लोर 1st मेन रोड गांधी नगर~इक्कीस बटा दो सेकंड फ्लोर फ़र्स्ट मेन रोड गांधी नगर +नंबर २२/१८ ३rd फ्लोर सराय बोउ अली शू मार्केट~नंबर बाईस बटा अठारह थर्ड फ्लोर सराय बोउ अली शू मार्केट +14/3, मथुरा रोड~चौदह बटा तीन, मथुरा रोड +यूनिट ३ १st फ्लोर नंबर ३७ सोलेमान खतर स्ट्रीट~यूनिट तीन फ़र्स्ट फ्लोर नंबर सैंतीस सोलेमान खतर स्ट्रीट +1st फ्लोर नंबर 52 नॉर्थ अबूज़र स्ट्रीट खान ए अंसारी स्ट्रीट शरीयती स्ट्रीट 16617~फ़र्स्ट फ्लोर नंबर बावन नॉर्थ अबूज़र स्ट्रीट खान ए अंसारी स्ट्रीट शरीयती स्ट्रीट एक छह छह एक सात +२०६ जय कॉम कॉम्प्लेक्स १st पोखरन रोड~दो सौ छह जय कॉम कॉम्प्लेक्स फ़र्स्ट पोखरन रोड +नंबर 36 2nd फ्लोर सुपर 8 फेज 1 एकबतन टाउन तेहरान 13947~नंबर छत्तीस सेकंड फ्लोर सुपर आठ फेज एक एकबतन टाउन तेहरान एक तीन नौ चार सात +२nd फ्लोर नंबर ८०८ आजादी स्ट्रीट~सेकंड फ्लोर नंबर आठ सौ आठ आजादी स्ट्रीट +2nd फ्लोर नंबर 15 बिफ़ोर कांदि स्ट्रीट नॉर्थ सोहरावर्दी स्ट्रीट 15669~सेकंड फ्लोर नंबर पंद्रह बिफ़ोर कांदि स्ट्रीट नॉर्थ सोहरावर्दी स्ट्रीट एक पाँच छह छह नौ +यूनिट ४ नंबर २५ २nd गोलहा स्ट्रीट काशनी स्ट्रीट नूर स्क्वेर~यूनिट चार नंबर पच्चीस सेकंड गोलहा स्ट्रीट काशनी स्ट्रीट नूर स्क्वेर +ईस्ट 3rd फ्लोर नंबर 70 नेक्स्ट दो तोहीद इंस्टीट्यूट परचम स्ट्रीट~ईस्ट थर्ड फ्लोर नंबर सत्तर नेक्स्ट दो तोहीद इंस्टीट्यूट परचम स्ट्रीट ३rd फ्लोर नंबर ५ हमेदन एली अपोज़िट लाले पार्क नॉर्थ कारगर स्ट्रीट~थर्ड फ्लोर नंबर पाँच हमेदन एली अपोज़िट लाले पार्क नॉर्थ कारगर स्ट्रीट 4th फ्लोर नंबर 1124 जमहोरी स्ट्रीट~फ़ोर्थ फ्लोर नंबर एक एक दो चार जमहोरी स्ट्रीट ५th फ्लोर नंबर ७/१ १३th एली शाहिद अराबली स्ट्रीट~फ़िफ्थ फ्लोर नंबर सात बटा एक थर्टींथ एली शाहिद अराबली स्ट्रीट -11, 80 फीट रोड, इंडियन ऑयल पेट्रोल पंप, कोरमंगला 6th ब्लॉक, बैंगलोर के सामने~एक एक, आठ शून्य फीट रोड, इंडियन ऑयल पेट्रोल पंप, कोरमंगला सिक्स्थ ब्लॉक, बैंगलोर के सामने -२१/११, जे ब्लॉक, ६th एवेन्यू मेन रोड, अन्ना नगर पूर्व, चेन्नई~दो एक बटा एक एक, जे ब्लॉक, सिक्स्थ एवेन्यू मेन रोड, अन्ना नगर पूर्व, चेन्नई -32A नाज़ प्लाज़ा मेरिस रोड~तीन दो ए नाज़ प्लाज़ा मेरिस रोड -२१४ बी गोविंद पूरी स्ट्रीट नंबर २~दो एक चार बी गोविंद पूरी स्ट्रीट नंबर दो -4362 16वीं एवेन्यू एसडब्ल्यू, देवदार रैपिड्स, आई ए 52404~चार तीन छह दो सोलहवीं एवेन्यू एसडब्ल्यू, देवदार रैपिड्स, आई ए बावन हज़ार चार सौ चार +11, 80 फीट रोड, इंडियन ऑयल पेट्रोल पंप, कोरमंगला 6th ब्लॉक, बैंगलोर के सामने~ग्यारह, अस्सी फीट रोड, इंडियन ऑयल पेट्रोल पंप, कोरमंगला सिक्स्थ ब्लॉक, बैंगलोर के सामने +२१/११, जे ब्लॉक, ६th एवेन्यू मेन रोड, अन्ना नगर पूर्व, चेन्नई~इक्कीस बटा ग्यारह, जे ब्लॉक, सिक्स्थ एवेन्यू मेन रोड, अन्ना नगर पूर्व, चेन्नई +32A नाज़ प्लाज़ा मेरिस रोड~बत्तीस ए नाज़ प्लाज़ा मेरिस रोड +२१४ बी गोविंद पूरी स्ट्रीट नंबर २~दो सौ चौदह बी गोविंद पूरी स्ट्रीट नंबर दो +२५१३ ५३ एवेन्यू, मुंबई, महाराष्ट्र ४००००१~दो पाँच एक तीन तिरेपन एवेन्यू, मुंबई, महाराष्ट्र चार शून्य शून्य शून्य शून्य एक अमरावती ६५५९३०~अमरावती छह पाँच पाँच नौ तीन शून्य शिमला, हिमाचल प्रदेश 593988~शिमला, हिमाचल प्रदेश पाँच नौ तीन नौ आठ आठ -२७०४४० डॉसन आर डी, अल्बानी, जीए ३१७०७~दो सात शून्य चार चार शून्य डॉसन आर डी, अल्बानी, जीए तीन एक सात शून्य सात रांची, झारखंड 736557~रांची, झारखंड सात तीन छह पाँच पाँच सात कोहिमा, नागालैंड ४४८३७७~कोहिमा, नागालैंड चार चार आठ तीन सात सात मुंबई, महाराष्ट्र 839488~मुंबई, महाराष्ट्र आठ तीन नौ चार आठ आठ diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_cardinal.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_cardinal.txt index d607992d7..050310f9f 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_cardinal.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_cardinal.txt @@ -148,3 +148,14 @@ ०७३~शून्य सात तीन 0001~शून्य शून्य शून्य एक ०००~शून्य शून्य शून्य +3,24,50,000~तीन करोड़ चौबीस लाख पचास हज़ार +२,१२,१५,०००~दो करोड़ बारह लाख पंद्रह हज़ार +32,450,000~तीन करोड़ चौबीस लाख पचास हज़ार +४,९९,९९,०००~चार करोड़ निन्यानबे लाख निन्यानबे हज़ार +5,50,00,000~पाँच करोड़ पचास लाख +32,45,000~बत्तीस लाख पैंतालीस हज़ार +५,५६,३२०~पाँच लाख छप्पन हज़ार तीन सौ बीस +1,23,456~एक लाख तेईस हज़ार चार सौ छप्पन +12,345~बारह हज़ार तीन सौ पैंतालीस +११,२२०~ग्यारह हज़ार दो सौ बीस +1,00,00,000~एक करोड़ \ No newline at end of file diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_decimal.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_decimal.txt index 3582aff50..03b01de2f 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_decimal.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_decimal.txt @@ -18,3 +18,7 @@ १०००००००००००००.०००३~एक नील दशमलव शून्य शून्य शून्य तीन 1000000000000000.008~एक पद्म दशमलव शून्य शून्य आठ १०००००००००००००००००.४१२~एक शंख दशमलव चार एक दो +१९२.१६८~एक सौ बानबे दशमलव एक छह आठ +192.168~एक सौ बानबे दशमलव एक छह आठ +99.99~निन्यानबे दशमलव नौ नौ +९९.९९~निन्यानबे दशमलव नौ नौ diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_electronic.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_electronic.txt index fd1bf459d..3265724a3 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_electronic.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_electronic.txt @@ -1,50 +1,50 @@ -gmail.com~जीमेल डॉट कॉम -yahoo.com~याहू डॉट कॉम -hotmail.com~हॉटमेल डॉट कॉम -google.com~गूगल डॉट कॉम -kumaar.org~के यू एम ए ए आर डॉट ऑर्ग -kumaar.info~के यू एम ए ए आर डॉट इन्फो -kumar@gmail.com~के यू एम ए आर एट जीमेल डॉट कॉम -robin@hotmail.com~आर ओ बी आई एन एट हॉटमेल डॉट कॉम -kapil@live.com~के ए पी आई एल एट लाइव डॉट कॉम -sneha@live.com~एस एन ई एच ए एट लाइव डॉट कॉम -mayank@google.com~एम ए वाई ए एन के एट गूगल डॉट कॉम -charu@yahoo.com~सी एच ए आर यू एट याहू डॉट कॉम -john20@yahoo.com~जे ओ एच एन दो शून्य एट याहू डॉट कॉम -vivaan62@gmail.com~वी आई वी ए ए एन छह दो एट जीमेल डॉट कॉम -viaan15@kumaar.com~वी आई ए ए एन एक पाँच एट के यू एम ए ए आर डॉट कॉम -ltaa12@gmail.com~एल टी ए ए एक दो एट जीमेल डॉट कॉम -kristen11@hotmail.com~के आर आई एस टी ई एन एक एक एट हॉटमेल डॉट कॉम -dsmith@yahoo.com~डी एस एम आई टी एच एट याहू डॉट कॉम -hgarza@gmail.com~एच जी ए आर ज़ेड ए एट जीमेल डॉट कॉम -qhill@yahoo.com~क्यू एच आई एल एल एट याहू डॉट कॉम -green-turner.org~जी आर ई ई एन हाइफ़न टी यू आर एन ई आर डॉट ऑर्ग -sharma-badami.com~एस एच ए आर एम ए हाइफ़न बी ए डी ए एम आई डॉट कॉम -osborne-gross.com~ओ एस बी ओ आर एन ई हाइफ़न जी आर ओ एस एस डॉट कॉम -lucero-stevenson.net~एल यू सी ई आर ओ हाइफ़न एस टी ई वी ई एन एस ओ एन डॉट नेट -https://google.com/~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश गूगल डॉट कॉम फॉरवर्ड स्लैश -https://github.com/~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश गिटहब डॉट कॉम फॉरवर्ड स्लैश -https://wikipedia.org/~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश विकिपीडिया डॉट ऑर्ग फॉरवर्ड स्लैश -https://amazon.com/~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश अमेज़ोन डॉट कॉम फॉरवर्ड स्लैश -www.google.com~डब्ल्यू डब्ल्यू डब्ल्यू डॉट गूगल डॉट कॉम -https://www.ndtv.com~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश डब्ल्यू डब्ल्यू डब्ल्यू डॉट एन डी टी वी डॉट कॉम -https://www.rbi.org.in/~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश डब्ल्यू डब्ल्यू डब्ल्यू डॉट आर बी आई डॉट ऑर्ग डॉट इन फॉरवर्ड स्लैश -https://www.amity.edu~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश डब्ल्यू डब्ल्यू डब्ल्यू डॉट ए एम आई टी वाई डॉट ई डी यू -https://example.com/blog/~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश ई एक्स ए एम पी एल ई डॉट कॉम फॉरवर्ड स्लैश ब्लॉग फॉरवर्ड स्लैश -https://example.com/about.html~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश ई एक्स ए एम पी एल ई डॉट कॉम फॉरवर्ड स्लैश अबाउट डॉट एच टी एम एल -https://example.com/search.php~एच टी टी पी एस कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश ई एक्स ए एम पी एल ई डॉट कॉम फॉरवर्ड स्लैश सर्च डॉट पी एच पी -http://ati.edu~एच टी टी पी कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश ए टी आई डॉट ई डी यू -http://gcu.edu~एच टी टी पी कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश जी सी यू डॉट ई डी यू -http://pima.edu~एच टी टी पी कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश पी आई एम ए डॉट ई डी यू -bamu.nic.in/~बी ए एम यू डॉट एन आई सी डॉट इन फॉरवर्ड स्लैश -bieap.gov.in/~बी आई ई ए पी डॉट जी ओ वी डॉट इन फॉरवर्ड स्लैश -www.sharda.ac.in~डब्ल्यू डब्ल्यू डब्ल्यू डॉट एस एच ए आर डी ए डॉट ए सी डॉट इन -C:\Users\HP\Desktop\~सी कोलन बैकवर्ड स्लैश यू एस ई आर एस बैकवर्ड स्लैश एच पी बैकवर्ड स्लैश डी ई एस के टी ओ पी बैकवर्ड स्लैश -C:\Users\HP\Downloads\~सी कोलन बैकवर्ड स्लैश यू एस ई आर एस बैकवर्ड स्लैश एच पी बैकवर्ड स्लैश डी ओ डब्ल्यू एन एल ओ ए डी एस बैकवर्ड स्लैश -C:\Users\HP\Documents\Zoom~सी कोलन बैकवर्ड स्लैश यू एस ई आर एस बैकवर्ड स्लैश एच पी बैकवर्ड स्लैश डी ओ सी यू एम ई एन टी एस बैकवर्ड स्लैश ज़ेड ओ ओ एम -/home/desktop~फॉरवर्ड स्लैश होम फॉरवर्ड स्लैश डेस्कटॉप -/etc/apache~फॉरवर्ड स्लैश ई टी सी फॉरवर्ड स्लैश ए पी ए सी एच ई -/var/www~फॉरवर्ड स्लैश वी ए आर फॉरवर्ड स्लैश डब्ल्यू डब्ल्यू डब्ल्यू +gmail.com~gmail डॉट com +yahoo.com~yahoo डॉट com +hotmail.com~hotmail डॉट com +google.com~google डॉट com +kumaar.org~kumaar डॉट org +kumaar.info~kumaar डॉट info +kumar@gmail.com~kumar एट gmail डॉट com +robin@hotmail.com~robin एट hotmail डॉट com +kapil@live.com~kapil एट live डॉट com +sneha@live.com~sneha एट live डॉट com +mayank@google.com~mayank एट google डॉट com +charu@yahoo.com~charu एट yahoo डॉट com +john20@yahoo.com~john दो शून्य एट yahoo डॉट com +vivaan62@gmail.com~vivaan छह दो एट gmail डॉट com +viaan15@kumaar.com~viaan एक पाँच एट kumaar डॉट com +ltaa12@gmail.com~ltaa एक दो एट gmail डॉट com +kristen11@hotmail.com~kristen एक एक एट hotmail डॉट com +dsmith@yahoo.com~dsmith एट yahoo डॉट com +hgarza@gmail.com~hgarza एट gmail डॉट com +qhill@yahoo.com~qhill एट yahoo डॉट com +green-turner.org~green हाइफ़न turner डॉट org +sharma-badami.com~sharma हाइफ़न badami डॉट com +osborne-gross.com~osborne हाइफ़न gross डॉट com +lucero-stevenson.net~lucero हाइफ़न stevenson डॉट net +https://google.com/~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश google डॉट com फॉरवर्ड स्लैश +https://github.com/~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश github डॉट com फॉरवर्ड स्लैश +https://wikipedia.org/~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश wikipedia डॉट org फॉरवर्ड स्लैश +https://amazon.com/~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश amazon डॉट com फॉरवर्ड स्लैश +www.google.com~www डॉट google डॉट com +https://www.ndtv.com~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश www डॉट ndtv डॉट com +https://www.rbi.org.in/~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश www डॉट rbi डॉट org डॉट in फॉरवर्ड स्लैश +https://www.amity.edu~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश www डॉट amity डॉट edu +https://example.com/blog/~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश example डॉट com फॉरवर्ड स्लैश blog फॉरवर्ड स्लैश +https://example.com/about.html~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश example डॉट com फॉरवर्ड स्लैश about डॉट html +https://example.com/search.php~https कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश example डॉट com फॉरवर्ड स्लैश search डॉट php +http://ati.edu~http कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश ati डॉट edu +http://gcu.edu~http कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश gcu डॉट edu +http://pima.edu~http कोलन फॉरवर्ड स्लैश फॉरवर्ड स्लैश pima डॉट edu +bamu.nic.in/~bamu डॉट nic डॉट in फॉरवर्ड स्लैश +bieap.gov.in/~bieap डॉट gov डॉट in फॉरवर्ड स्लैश +www.sharda.ac.in~www डॉट sharda डॉट ac डॉट in +C:\Users\HP\Desktop~C कोलन बैकवर्ड स्लैश Users बैकवर्ड स्लैश HP बैकवर्ड स्लैश Desktop +C:\Users\HP\Downloads~C कोलन बैकवर्ड स्लैश Users बैकवर्ड स्लैश HP बैकवर्ड स्लैश Downloads +C:\Users\HP\Documents\Zoom~C कोलन बैकवर्ड स्लैश Users बैकवर्ड स्लैश HP बैकवर्ड स्लैश Documents बैकवर्ड स्लैश Zoom +/home/desktop~फॉरवर्ड स्लैश home फॉरवर्ड स्लैश desktop +/etc/apache~फॉरवर्ड स्लैश etc फॉरवर्ड स्लैश apache +/var/www~फॉरवर्ड स्लैश var फॉरवर्ड स्लैश www 192.168.1.1~एक नौ दो डॉट एक छह आठ डॉट एक डॉट एक 10.0.0.1~एक शून्य डॉट शून्य डॉट शून्य डॉट एक 83.54.245.61~आठ तीन डॉट पाँच चार डॉट दो चार पाँच डॉट छह एक @@ -55,11 +55,11 @@ C:\Users\HP\Documents\Zoom~सी कोलन बैकवर्ड स्ल आईपी एड्रेस 10.0.0.1~आईपी एड्रेस एक शून्य डॉट शून्य डॉट शून्य डॉट एक ip address 192.168.1.1~ip address एक नौ दो डॉट एक छह आठ डॉट एक डॉट एक ip address 10.0.0.1~ip address एक शून्य डॉट शून्य डॉट शून्य डॉट एक -report.pdf~आर ई पी ओ आर टी डॉट पी डी एफ -photo.jpg~पी एच ओ टी ओ डॉट जे पी जी -data.csv~डेटा डॉट सी एस वी -robinson.org~आर ओ बी आई एन एस ओ एन डॉट ऑर्ग -anand@gmail.com~ए एन ए एन डी एट जीमेल डॉट कॉम +report.pdf~report डॉट pdf +photo.jpg~photo डॉट jpg +data.csv~data डॉट csv +robinson.org~robinson डॉट org +anand@gmail.com~anand एट gmail डॉट com Al₂(SO₄)₃~ए एल दो ओपन ब्रेकेट एस ओ चार क्लोज़ ब्रेकेट तीन C₂H₄~सी दो एच चार -home/desktop~होम फॉरवर्ड स्लैश डेस्कटॉप \ No newline at end of file +home/desktop~home फॉरवर्ड स्लैश desktop \ No newline at end of file diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_money.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_money.txt index 26b292a17..e5f157872 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_money.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_money.txt @@ -125,4 +125,20 @@ $1.2000~एक डॉलर बीस सेंट ₹5.00~पाँच रुपए ₹१~एक रुपया ₹२.१२३~दो दशमलव एक दो तीन रुपए -₹१.१२३४~एक दशमलव एक दो तीन चार रुपए \ No newline at end of file +₹१.१२३४~एक दशमलव एक दो तीन चार रुपए +₹3,24,50,000~तीन करोड़ चौबीस लाख पचास हज़ार रुपए +₹32,450,000~तीन करोड़ चौबीस लाख पचास हज़ार रुपए +₹5,50,00,000~पाँच करोड़ पचास लाख रुपए +₹12,54,000~बारह लाख चौवन हज़ार रुपए +₹1,00,000~एक लाख रुपए +₹2,148~दो हज़ार एक सौ अड़तालीस रुपए +₹99,999~निन्यानबे हज़ार नौ सौ निन्यानबे रुपए +₹३,२४,५०,०००~तीन करोड़ चौबीस लाख पचास हज़ार रुपए +₹३२,४५०,०००~तीन करोड़ चौबीस लाख पचास हज़ार रुपए +₹२,१२,१५,०००~दो करोड़ बारह लाख पंद्रह हज़ार रुपए +₹५,५०,००,०००~पाँच करोड़ पचास लाख रुपए +₹१२,५४,०००~बारह लाख चौवन हज़ार रुपए +₹५,५६,३२०~पाँच लाख छप्पन हज़ार तीन सौ बीस रुपए +₹१,००,०००~एक लाख रुपए +₹२,१४८~दो हज़ार एक सौ अड़तालीस रुपए +₹९९,९९९~निन्यानबे हज़ार नौ सौ निन्यानबे रुपए \ No newline at end of file diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_roman.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_roman.txt index 00f697a89..340c754ed 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_roman.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_roman.txt @@ -12,7 +12,7 @@ राष्ट्रीय राजमार्ग-IV~राष्ट्रीय राजमार्ग चार रोहिणी आर एस-I~रोहिणी आर एस एक पीएसएलवी सी-IV~पीएसएलवी सी चार -ISRO मिशन-III~आई एस आर ओ मिशन तीन +ISRO मिशन-III~ISRO मिशन तीन कक्षा XII की परीक्षा~कक्षा बारह की परीक्षा XIIवीं कक्षा की परीक्षा~बारहवीं कक्षा की परीक्षा भाग II का सारांश~भाग दो का सारांश diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_serial.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_serial.txt index 4c3880fb9..4a105554d 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_serial.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_serial.txt @@ -18,4 +18,14 @@ 10-20-30~दस-बीस-तीस 1-800-999~एक-आठ सौ-नौ सौ निन्यानबे पृथ्वी-4~पृथ्वी-चार -ब्रह्मोस-1~ब्रह्मोस-एक \ No newline at end of file +ब्रह्मोस-1~ब्रह्मोस-एक +Q1~क्यू एक +A10~ए दस +A12~ए बारह +B-60~बी-साठ +ABC-123~ए बी सी-एक सौ तेईस +FY2024~एफ वाई दो शून्य दो चार +H2O~एच दो ओ +CO2~सी ओ दो +ABCDE1234F~ए बी सी डी ई एक दो तीन चार एफ +F16~एफ सोलह \ No newline at end of file diff --git a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_word.txt b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_word.txt index e9649919e..e7a284e9f 100644 --- a/tests/nemo_text_processing/hi/data_text_normalization/test_cases_word.txt +++ b/tests/nemo_text_processing/hi/data_text_normalization/test_cases_word.txt @@ -13,4 +13,8 @@ टाटा~टाटा ~ झ~झ -संगीत~संगीत \ No newline at end of file +संगीत~संगीत +This is a sentence.~This is a sentence. +google~google +mera email hai~mera email hai +user@~user@ \ No newline at end of file