diff --git a/nemo_text_processing/__init__.py b/nemo_text_processing/__init__.py index bc443be41..4fc25d0d3 100644 --- a/nemo_text_processing/__init__.py +++ b/nemo_text_processing/__init__.py @@ -1,4 +1,4 @@ -# Copyright (c) 2021, NVIDIA CORPORATION. All rights reserved. +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. # # Licensed under the Apache License, Version 2.0 (the "License"); # you may not use this file except in compliance with the License. diff --git a/nemo_text_processing/inverse_text_normalization/__init__.py b/nemo_text_processing/inverse_text_normalization/__init__.py index 72d9c125c..4fc25d0d3 100644 --- a/nemo_text_processing/inverse_text_normalization/__init__.py +++ b/nemo_text_processing/inverse_text_normalization/__init__.py @@ -1,4 +1,4 @@ -# Copyright (c) 2021, NVIDIA CORPORATION. All rights reserved. +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. # # Licensed under the Apache License, Version 2.0 (the "License"); # you may not use this file except in compliance with the License. @@ -11,5 +11,3 @@ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. # See the License for the specific language governing permissions and # limitations under the License. - -from nemo_text_processing.inverse_text_normalization.inverse_normalize import InverseNormalizer diff --git a/nemo_text_processing/inverse_text_normalization/inverse_normalize.py b/nemo_text_processing/inverse_text_normalization/inverse_normalize.py index 9a6fcc64c..3993f230c 100644 --- a/nemo_text_processing/inverse_text_normalization/inverse_normalize.py +++ b/nemo_text_processing/inverse_text_normalization/inverse_normalize.py @@ -1,4 +1,4 @@ -# Copyright (c) 2021, NVIDIA CORPORATION. All rights reserved. +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. # # Licensed under the Apache License, Version 2.0 (the "License"); # you may not use this file except in compliance with the License. @@ -146,12 +146,18 @@ def __init__( from nemo_text_processing.inverse_text_normalization.ko.verbalizers.verbalize_final import ( VerbalizeFinalFst, ) + elif lang == 'ta': # Tamil + from nemo_text_processing.inverse_text_normalization.ta.taggers.tokenize_and_classify import ClassifyFst + from nemo_text_processing.inverse_text_normalization.ta.verbalizers.verbalize_final import ( + VerbalizeFinalFst, + ) else: raise NotImplementedError(f"Language {lang} has not been supported yet.") self.tagger = ClassifyFst( cache_dir=cache_dir, whitelist=whitelist, overwrite_cache=overwrite_cache, input_case=input_case ) + self.verbalizer = VerbalizeFinalFst() self.parser = TokenParser() self.lang = lang @@ -211,6 +217,7 @@ def parse_args(): 'mr', 'ja', 'ko', + 'ta', ], default="en", type=str, diff --git a/nemo_text_processing/inverse_text_normalization/run_evaluate.py b/nemo_text_processing/inverse_text_normalization/run_evaluate.py index cf9b29fce..643dad7b1 100644 --- a/nemo_text_processing/inverse_text_normalization/run_evaluate.py +++ b/nemo_text_processing/inverse_text_normalization/run_evaluate.py @@ -1,4 +1,4 @@ -# Copyright (c) 2024, NVIDIA CORPORATION. All rights reserved. +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. # # Licensed under the Apache License, Version 2.0 (the "License"); # you may not use this file except in compliance with the License. @@ -46,6 +46,7 @@ def parse_args(): "hi_en", "hy", "ko", + "ta", "mr", "pt", "ru", @@ -78,9 +79,13 @@ def parse_args(): if args.lang == 'en': from nemo_text_processing.inverse_text_normalization.en.clean_eval_data import filter_loaded_data file_path = args.input - inverse_normalizer = InverseNormalizer(lang=args.lang, input_case=args.input_case) - print("Loading training data: " + file_path) + inverse_normalizer = InverseNormalizer( + lang=args.lang, + input_case=args.input_case, + overwrite_cache=True, + ) + if args.output_case == "lower_cased": to_lower = True elif args.output_case == "cased": @@ -96,9 +101,9 @@ def parse_args(): sentences_un_normalized, sentences_normalized, _ = training_data_to_sentences(training_data) print("- Data: " + str(len(sentences_normalized)) + " sentences") sentences_prediction = inverse_normalizer.inverse_normalize_list(sentences_normalized) - with open('result.log', 'w') as ofp: + with open("result.log", "w", encoding="utf-8") as ofp: for inp, out in zip(sentences_un_normalized, sentences_prediction): - ofp.write(f'{inp==out}; {inp}\t{out}\n') + ofp.write(f"{inp == out}; {inp}\t{out}\n") print("- Denormalized. Evaluating...") sentences_accuracy = evaluate( @@ -108,21 +113,31 @@ def parse_args(): print("Token level evaluation...") tokens_per_type = training_data_to_tokens(training_data, category=args.category) + token_accuracy = {} + for token_type in tokens_per_type: print("- Token type: " + token_type) tokens_un_normalized, tokens_normalized = tokens_per_type[token_type] print(" - Data: " + str(len(tokens_normalized)) + " tokens") + tokens_prediction = inverse_normalizer.inverse_normalize_list(tokens_normalized) + print(" - Denormalized. Evaluating...") - token_accuracy[token_type] = evaluate(tokens_prediction, tokens_un_normalized, input=tokens_normalized) + token_accuracy[token_type] = evaluate( + tokens_prediction, + tokens_un_normalized, + input=tokens_normalized, + ) print(" - Accuracy: " + str(token_accuracy[token_type])) + token_count_per_type = {token_type: len(tokens_per_type[token_type][0]) for token_type in tokens_per_type} + token_weighted_accuracy = [ token_count_per_type[token_type] * accuracy for token_type, accuracy in token_accuracy.items() ] - print("- Accuracy: " + str(sum(token_weighted_accuracy) / sum(token_count_per_type.values()))) + print("- Accuracy: " + str(sum(token_weighted_accuracy) / sum(token_count_per_type.values()))) print(" - Total: " + str(sum(token_count_per_type.values())), '\n') for token_type in token_accuracy: diff --git a/nemo_text_processing/inverse_text_normalization/ta/data/__init__.py b/nemo_text_processing/inverse_text_normalization/ta/data/__init__.py new file mode 100644 index 000000000..4fc25d0d3 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/ta/data/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/inverse_text_normalization/ta/data/numbers/__init__.py b/nemo_text_processing/inverse_text_normalization/ta/data/numbers/__init__.py new file mode 100644 index 000000000..4fc25d0d3 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/ta/data/numbers/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/inverse_text_normalization/ta/data/numbers/digit.tsv b/nemo_text_processing/inverse_text_normalization/ta/data/numbers/digit.tsv new file mode 100644 index 000000000..f7c5e1f86 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/ta/data/numbers/digit.tsv @@ -0,0 +1,10 @@ +௧ ஒன்று +௧ ஒரு +௨ இரண்டு +௩ மூன்று +௪ நான்கு +௫ ஐந்து +௬ ஆறு +௭ ஏழு +௮ எட்டு +௯ ஒன்பது \ No newline at end of file diff --git a/nemo_text_processing/inverse_text_normalization/ta/data/numbers/hundreds.tsv b/nemo_text_processing/inverse_text_normalization/ta/data/numbers/hundreds.tsv new file mode 100644 index 000000000..a217d3fb0 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/ta/data/numbers/hundreds.tsv @@ -0,0 +1,11 @@ +௧ நூறு +௨ இருநூறு +௩ முன்னூறு +௪ நானூறு +௫ ஐநூறு +௬ அறுநூறு +௭ எழுநூறு +௮ எண்ணூறு +௮ எட்டுநூறு +௯ தொள்ளுநூறு +௯ தொள்ளாயிரம் diff --git a/nemo_text_processing/inverse_text_normalization/ta/data/numbers/teens_and_ties.tsv b/nemo_text_processing/inverse_text_normalization/ta/data/numbers/teens_and_ties.tsv new file mode 100644 index 000000000..a700e3138 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/ta/data/numbers/teens_and_ties.tsv @@ -0,0 +1,26 @@ +௧௦ பத்து +௧௧ பதினொன்று +௧௨ பன்னிரண்டு +௧௩ பதின்மூன்று +௧௪ பதினான்கு +௧௫ பதினைந்து +௧௬ பதினாறு +௧௭ பதினேழு +௧௮ பதினெட்டு +௧௯ பத்தொன்பது +௨௦ இருபது +௨ இருபத்து +௩௦ முப்பது +௩ முப்பத்து +௪௦ நாற்பது +௪ நாற்பத்து +௫௦ ஐம்பது +௫ ஐம்பத்து +௬௦ அறுபது +௬ அறுபத்து +௭௦ எழுபது +௭ எழுபத்து +௮௦ எண்பது +௮ எண்பத்து +௯௦ தொண்ணூறு +௯ தொண்ணூற்று \ No newline at end of file diff --git a/nemo_text_processing/inverse_text_normalization/ta/data/numbers/thousands.tsv b/nemo_text_processing/inverse_text_normalization/ta/data/numbers/thousands.tsv new file mode 100644 index 000000000..04e932ab2 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/ta/data/numbers/thousands.tsv @@ -0,0 +1,10 @@ +௧ ஆயிரம் +௧ ஒன்றாயிரம் +௨ இரண்டாயிரம் +௩ மூன்றாயிரம் +௪ நான்காயிரம் +௫ ஐந்தாயிரம் +௬ ஆறாயிரம் +௭ ஏழாயிரம் +௮ எட்டாயிரம் +௯ ஒன்பதாயிரம் diff --git a/nemo_text_processing/inverse_text_normalization/ta/data/numbers/zero.tsv b/nemo_text_processing/inverse_text_normalization/ta/data/numbers/zero.tsv new file mode 100644 index 000000000..77c135fbb --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/ta/data/numbers/zero.tsv @@ -0,0 +1 @@ +௦ சுழியம் diff --git a/nemo_text_processing/inverse_text_normalization/ta/graph_utils.py b/nemo_text_processing/inverse_text_normalization/ta/graph_utils.py new file mode 100644 index 000000000..fc6f652b0 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/ta/graph_utils.py @@ -0,0 +1,197 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# Copyright 2013 and onwards Google, Inc. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import logging +import os +import string +from pathlib import Path +from typing import Dict + +import pynini +from pynini import Far +from pynini.examples import plurals +from pynini.export import export +from pynini.lib import byte, pynutil, utf8 + +from nemo_text_processing.inverse_text_normalization.ta.utils import get_abs_path, load_labels + +NEMO_CHAR = utf8.VALID_UTF8_CHAR + +graph_digit = pynini.string_file(get_abs_path("data/numbers/digit.tsv")) + +NEMO_HEX = pynini.union(*string.hexdigits).optimize() +NEMO_NON_BREAKING_SPACE = u"\u00a0" +NEMO_ZWNJ = u"\u200c" +NEMO_SPACE = " " +NEMO_WHITE_SPACE = pynini.union(" ", "\t", "\n", "\r", u"\u00a0").optimize() +NEMO_NOT_SPACE = pynini.difference(NEMO_CHAR, NEMO_WHITE_SPACE).optimize() +NEMO_NOT_QUOTE = pynini.difference(NEMO_CHAR, r'"').optimize() + +NEMO_PUNCT = pynini.union(*map(pynini.escape, string.punctuation)).optimize() +NEMO_GRAPH = pynini.union(NEMO_CHAR, NEMO_PUNCT).optimize() + +NEMO_SIGMA = pynini.closure(NEMO_CHAR) + +delete_space = pynutil.delete(pynini.closure(NEMO_WHITE_SPACE)) +delete_zero_or_one_space = pynutil.delete(pynini.closure(NEMO_WHITE_SPACE, 0, 1)) +insert_space = pynutil.insert(" ") +delete_extra_space = pynini.cross(pynini.closure(NEMO_WHITE_SPACE, 1), " ") +delete_preserve_order = pynini.closure( + pynutil.delete(" preserve_order: true") + | (pynutil.delete(" field_order: \"") + NEMO_NOT_QUOTE + pynutil.delete("\"")) +) + + +MIN_NEG_WEIGHT = -0.0001 +MIN_POS_WEIGHT = 0.0001 +INPUT_CASED = "cased" +INPUT_LOWER_CASED = "lower_cased" +MINUS = pynini.union("மைனஸ்", "எதிர்மறை").optimize() + + +def generator_main(file_name: str, graphs: Dict[str, 'pynini.FstLike']): + """ + Exports graph as OpenFst finite state archive (FAR) file with given file name and rule name. + + Args: + file_name: exported file name + graphs: Mapping of a rule name and Pynini WFST graph to be exported + """ + exporter = export.Exporter(file_name) + for rule, graph in graphs.items(): + exporter[rule] = graph.optimize() + exporter.close() + logging.info(f'Created {file_name}') + + +def convert_space(fst) -> 'pynini.FstLike': + """ + Converts space to nonbreaking space. + Used only in tagger grammars for transducing token values within quotes, e.g. name: "hello kitty" + This is making transducer significantly slower, so only use when there could be potential spaces within quotes, otherwise leave it. + + Args: + fst: input fst + + Returns output fst where breaking spaces are converted to non breaking spaces + """ + return fst @ pynini.cdrewrite(pynini.cross(NEMO_SPACE, NEMO_NON_BREAKING_SPACE), "", "", NEMO_SIGMA) + + +def string_map_cased(input_file: str, input_case: str = INPUT_LOWER_CASED): + labels = load_labels(input_file) + + if input_case == INPUT_CASED: + additional_labels = [] + for written, spoken, *weight in labels: + written_capitalized = written[0].upper() + written[1:] + additional_labels.extend( + [ + [written_capitalized, spoken.capitalize()], # first letter capitalized + [ + written_capitalized, + spoken.upper().replace(" AND ", " and "), + ], # # add pairs with the all letters capitalized + ] + ) + + spoken_no_space = spoken.replace(" ", "") + # add abbreviations without spaces (both lower and upper case), i.e. "BMW" not "B M W" + if len(spoken) == (2 * len(spoken_no_space) - 1): + logging.debug(f"This is weight {weight}") + if len(weight) == 0: + additional_labels.extend( + [[written, spoken_no_space], [written_capitalized, spoken_no_space.upper()]] + ) + else: + additional_labels.extend( + [ + [written, spoken_no_space, weight[0]], + [written_capitalized, spoken_no_space.upper(), weight[0]], + ] + ) + labels += additional_labels + + whitelist = pynini.string_map(labels).invert().optimize() + return whitelist + + +class GraphFst: + """ + Base class for all grammar fsts. + + Args: + name: name of grammar class + kind: either 'classify' or 'verbalize' + deterministic: if True will provide a single transduction option, + for False multiple transduction are generated (used for audio-based normalization) + """ + + def __init__(self, name: str, kind: str, deterministic: bool = True): + self.name = name + self.kind = kind + self._fst = None + self.deterministic = deterministic + + self.far_path = Path(os.path.dirname(__file__) + '/grammars/' + kind + '/' + name + '.far') + if self.far_exist(): + self._fst = Far(self.far_path, mode="r", arc_type="standard", far_type="default").get_fst() + + def far_exist(self) -> bool: + """ + Returns true if FAR can be loaded + """ + return self.far_path.exists() + + @property + def fst(self) -> 'pynini.FstLike': + return self._fst + + @fst.setter + def fst(self, fst): + self._fst = fst + + def add_tokens(self, fst) -> 'pynini.FstLike': + """ + Wraps class name around to given fst + + Args: + fst: input fst + + Returns: + Fst: fst + """ + return pynutil.insert(f"{self.name} {{ ") + fst + pynutil.insert(" }") + + def delete_tokens(self, fst) -> 'pynini.FstLike': + """ + Deletes class name wrap around output of given fst + + Args: + fst: input fst + + Returns: + Fst: fst + """ + res = ( + pynutil.delete(f"{self.name}") + + delete_space + + pynutil.delete("{") + + delete_space + + fst + + delete_space + + pynutil.delete("}") + ) + return res @ pynini.cdrewrite(pynini.cross(u"\u00a0", " "), "", "", NEMO_SIGMA) diff --git a/nemo_text_processing/inverse_text_normalization/ta/taggers/__init__.py b/nemo_text_processing/inverse_text_normalization/ta/taggers/__init__.py new file mode 100644 index 000000000..53ede5f52 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/ta/taggers/__init__.py @@ -0,0 +1,14 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# Copyright 2024 and onwards Google, Inc. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/nemo_text_processing/inverse_text_normalization/ta/taggers/cardinal.py b/nemo_text_processing/inverse_text_normalization/ta/taggers/cardinal.py new file mode 100644 index 000000000..0f6b247f3 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/ta/taggers/cardinal.py @@ -0,0 +1,583 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +import nemo_text_processing.inverse_text_normalization.ta.utils +from nemo_text_processing.inverse_text_normalization.ta.graph_utils import NEMO_SIGMA, GraphFst, delete_space + + +class CardinalFst(GraphFst): + + def __init__(self): + super().__init__(name="cardinal", kind="classify") + + graph_zero_raw = pynini.string_file( + nemo_text_processing.inverse_text_normalization.ta.utils.get_abs_path("data/numbers/zero.tsv") + ) + graph_digit_raw = pynini.string_file( + nemo_text_processing.inverse_text_normalization.ta.utils.get_abs_path("data/numbers/digit.tsv") + ) + graph_teens_and_ties_raw = pynini.string_file( + nemo_text_processing.inverse_text_normalization.ta.utils.get_abs_path("data/numbers/teens_and_ties.tsv") + ) + graph_hundreds_raw = pynini.string_file( + nemo_text_processing.inverse_text_normalization.ta.utils.get_abs_path("data/numbers/hundreds.tsv") + ) + graph_thousands_raw = pynini.string_file( + nemo_text_processing.inverse_text_normalization.ta.utils.get_abs_path("data/numbers/thousands.tsv") + ) + + hundred_join_pairs = [] + + with open( + nemo_text_processing.inverse_text_normalization.ta.utils.get_abs_path("data/numbers/hundreds.tsv"), + encoding="utf-8-sig", + ) as f: + for line in f: + line = line.strip().lstrip("\ufeff") + if not line: + continue + + numeral, word = line.split("\t") + + if word.endswith("று"): + word = word.removesuffix("று") + "ற்று" + elif word.endswith("ம்"): + word = word.removesuffix("ம்") + "த்து" + + hundred_join_pairs.append((word, numeral)) + + graph_hundred_join = pynini.string_map(hundred_join_pairs) + + thousand_join_pairs = [] + + with open( + nemo_text_processing.inverse_text_normalization.ta.utils.get_abs_path("data/numbers/thousands.tsv"), + encoding="utf-8-sig", + ) as f: + for line in f: + line = line.strip().lstrip("\ufeff") + if not line: + continue + + value, word = line.split("\t") + + if word.endswith("ம்"): + join_word = word.removesuffix("ம்") + "த்து" + thousand_join_pairs.append((join_word, value)) + + graph_all_thousand_join = pynini.string_map(thousand_join_pairs) + + graph_zero = graph_zero_raw.copy().invert() + graph_digit = graph_digit_raw.copy().invert() + graph_teens_and_ties = graph_teens_and_ties_raw.copy().invert() + graph_hundreds = graph_hundreds_raw.copy().invert() + graph_thousands = graph_thousands_raw.copy().invert() + + # Numeric (input-side) digit graphs. + graph_numeric_digit = (graph_zero_raw.project("input") | graph_digit_raw.project("input")).optimize() + + graph_numeric_two_digit = (graph_numeric_digit + graph_numeric_digit).optimize() + + graph_numeric_three_digit = (graph_numeric_digit + graph_numeric_digit + graph_numeric_digit).optimize() + + # Two-digit composition. + graph_join_digit = graph_digit_raw.project("input").optimize() + + graph_ties_join = graph_teens_and_ties @ graph_join_digit + graph_two_digit_composed = graph_ties_join + delete_space + graph_digit + + fusion_rules = pynini.union( + pynini.cross("த்தை", "த்து ஐ"), + pynini.cross("த்தா", "த்து ஆ"), + pynini.cross("த்தே", "த்து ஏ"), + pynini.cross("த்தெ", "த்து எ"), + pynini.cross("த்தொ", "த்து ஒ"), + pynini.cross("ற்றா", "ற்று ஆ"), + ) + + fusion_rewrite = pynini.cdrewrite(fusion_rules, "", "", NEMO_SIGMA) + graph_two_digit_fused = fusion_rewrite @ graph_two_digit_composed + + self.graph_two_digit = (graph_teens_and_ties | graph_two_digit_composed | graph_two_digit_fused).optimize() + + # Accept both verbal and numeric Tamil digit forms. + graph_digit_any = (graph_digit | graph_numeric_digit).optimize() + + graph_two_digit_any = (self.graph_two_digit | graph_numeric_two_digit).optimize() + + graph_digit_any_with_zero = (pynutil.insert("௦") + graph_digit_any).optimize() + + graph_digit_any_any = (graph_digit_any | graph_digit_any_with_zero).optimize() + + # Single source of truth for the thousand-fusion pairs. + # The ஆயிரம் / ஆயிரத்து variants are derived programmatically. + thousand_fusion_pairs = [ + ("காயிரம்", "கு ஆயிரம்"), + ("ட்டாயிரம்", "ட்டு ஆயிரம்"), + ("றாயிரம்", "று ஆயிரம்"), + ("த்தாயிரம்", "த்து ஆயிரம்"), + ("ந்தாயிரம்", "ந்து ஆயிரம்"), + ] + + thousand_fusion_pairs_with_join = [] + + for fused_word, expanded_word in thousand_fusion_pairs: + thousand_fusion_pairs_with_join.append((fused_word, expanded_word)) + + fused_join_word = fused_word.removesuffix("ம்") + "த்து" + expanded_join_word = expanded_word.removesuffix("ம்") + "த்து" + + thousand_fusion_pairs_with_join.append((fused_join_word, expanded_join_word)) + + graph_thousand_fusion_rules = pynini.string_map(thousand_fusion_pairs_with_join) + + graph_thousand_fusion_rewrite = pynini.cdrewrite( + graph_thousand_fusion_rules, + "", + "", + NEMO_SIGMA, + ) + + graph_two_digit_thousand_multiplier = (self.graph_two_digit | graph_numeric_two_digit).optimize() + + graph_two_digit_thousand_exact = graph_thousand_fusion_rewrite @ ( + graph_two_digit_thousand_multiplier + pynini.cross(" ", "") + pynini.cross("ஆயிரம்", "௦௦௦") + ) + + # Hundreds. + self.graph_exact_hundreds = graph_hundreds + pynutil.insert("௦௦") + + graph_hundred_with_digit = graph_hundred_join + delete_space + graph_digit_any_with_zero + + graph_hundred_with_two_digit = graph_hundred_join + delete_space + graph_two_digit_any + + graph_hundred_with_numeric_two_digit = graph_hundred_join + delete_space + graph_numeric_two_digit + + self.graph_hundred_with_remainder = ( + graph_hundred_with_digit | graph_hundred_with_two_digit | graph_hundred_with_numeric_two_digit + ).optimize() + + graph_hundred_remainder_any = ( + self.graph_hundred_with_remainder | self.graph_exact_hundreds | graph_numeric_three_digit + ).optimize() + + graph_single_thousand_hundred_remainder = ( + graph_all_thousand_join + delete_space + graph_hundred_remainder_any + ).optimize() + + # Thousands. + self.graph_exact_thousands = ( + graph_thousands + pynutil.insert("௦௦௦") | graph_two_digit_thousand_exact + ).optimize() + + graph_thousand_with_digit = graph_all_thousand_join + pynutil.insert("௦௦") + delete_space + graph_digit_any_any + + graph_thousand_with_two_digit = ( + graph_all_thousand_join + pynutil.insert("௦") + delete_space + graph_two_digit_any + ) + + graph_thousand_with_hundred = graph_all_thousand_join + delete_space + self.graph_exact_hundreds + + graph_thousand_with_hundred_remainder = graph_all_thousand_join + delete_space + graph_hundred_remainder_any + + # Two-digit thousand forms — all use the shared rewrite graph. + graph_two_digit_thousand_with_digit = graph_thousand_fusion_rewrite @ ( + graph_two_digit_thousand_multiplier + + pynini.cross(" ", "") + + pynini.cross("ஆயிரத்து", "") + + pynutil.insert("௦௦") + + delete_space + + graph_digit_any_any + ) + + graph_two_digit_thousand_with_two_digit = graph_thousand_fusion_rewrite @ ( + graph_two_digit_thousand_multiplier + + pynini.cross(" ", "") + + pynini.cross("ஆயிரத்து", "") + + pynutil.insert("௦") + + delete_space + + graph_two_digit_any + ) + + graph_two_digit_thousand_with_hundred = graph_thousand_fusion_rewrite @ ( + graph_two_digit_thousand_multiplier + + pynini.cross(" ", "") + + pynini.cross("ஆயிரத்து", "") + + delete_space + + self.graph_exact_hundreds + ) + + graph_two_digit_thousand_with_hundred_remainder = graph_thousand_fusion_rewrite @ ( + graph_two_digit_thousand_multiplier + + pynini.cross(" ", "") + + pynini.cross("ஆயிரத்து", "") + + delete_space + + graph_hundred_remainder_any + ) + + # Split by thousand-multiplier width so padding zeros are correct. + graph_thousand_with_remainder_single = ( + graph_thousand_with_digit + | graph_thousand_with_two_digit + | graph_thousand_with_hundred + | graph_thousand_with_hundred_remainder + ).optimize() + + graph_thousand_with_remainder_two_digit = ( + graph_two_digit_thousand_with_digit + | graph_two_digit_thousand_with_two_digit + | graph_two_digit_thousand_with_hundred + | graph_two_digit_thousand_with_hundred_remainder + ).optimize() + + self.graph_thousand_with_remainder = ( + graph_thousand_with_remainder_single | graph_thousand_with_remainder_two_digit + ).optimize() + + # Lakh. + graph_lakh_word = pynini.union( + pynini.cross("இலட்சம்", ""), + pynini.cross("லட்சம்", ""), + ) + + graph_lakh_join_delete = pynini.union( + pynini.cross("இலட்சத்து", ""), + pynini.cross("லட்சத்து", ""), + ) + + graph_lakh_multiplier = (graph_digit_any | graph_two_digit_any).optimize() + + graph_single_lakh = graph_lakh_multiplier + delete_space + graph_lakh_word + pynutil.insert("௦௦௦௦௦") + + graph_two_digit_lakh = graph_two_digit_any + delete_space + graph_lakh_word + pynutil.insert("௦௦௦௦௦") + + self.graph_lakh = (graph_single_lakh | graph_two_digit_lakh).optimize() + + graph_lakh_with_digit = ( + graph_lakh_multiplier + + delete_space + + graph_lakh_join_delete + + pynutil.insert("௦௦௦௦") + + delete_space + + graph_digit_any + ) + + graph_lakh_with_two_digit = ( + graph_lakh_multiplier + + delete_space + + graph_lakh_join_delete + + pynutil.insert("௦௦௦") + + delete_space + + graph_two_digit_any + ) + + graph_lakh_with_hundred = ( + graph_lakh_multiplier + + delete_space + + graph_lakh_join_delete + + pynutil.insert("௦௦") + + delete_space + + self.graph_exact_hundreds + ) + + graph_lakh_with_hundred_remainder = ( + graph_lakh_multiplier + + delete_space + + graph_lakh_join_delete + + pynutil.insert("௦௦") + + delete_space + + graph_hundred_remainder_any + ) + + graph_lakh_single_with_thousand = ( + graph_digit_any + + delete_space + + graph_lakh_join_delete + + pynutil.insert("௦") + + delete_space + + graph_thousand_with_remainder_single + ) | ( + graph_digit_any + + delete_space + + graph_lakh_join_delete + + delete_space + + graph_thousand_with_remainder_two_digit + ) + + graph_lakh_two_digit_with_thousand = ( + graph_two_digit_any + + delete_space + + graph_lakh_join_delete + + pynutil.insert("௦") + + delete_space + + graph_thousand_with_remainder_single + ) | ( + graph_two_digit_any + + delete_space + + graph_lakh_join_delete + + delete_space + + graph_thousand_with_remainder_two_digit + ) + + graph_lakh_with_thousand = graph_lakh_single_with_thousand | graph_lakh_two_digit_with_thousand + + self.graph_lakh_with_remainder = ( + graph_lakh_with_digit + | graph_lakh_with_two_digit + | graph_lakh_with_hundred + | graph_lakh_with_hundred_remainder + | graph_lakh_with_thousand + ).optimize() + + # Crore. + graph_crore_word = pynini.union(pynini.cross("கோடி", "")) + + graph_single_crore = graph_digit_any + delete_space + graph_crore_word + pynutil.insert("௦௦௦௦௦௦௦") + + graph_two_digit_crore = graph_two_digit_any + delete_space + graph_crore_word + pynutil.insert("௦௦௦௦௦௦௦") + + self.graph_crore = (graph_single_crore | graph_two_digit_crore).optimize() + + graph_crore_join_delete = pynini.union(pynini.cross("கோடியே", "")) + + graph_crore_single_thousand_hundred_remainder = ( + graph_single_thousand_hundred_remainder + delete_space + pynini.cross("கோடியே", "") + ).optimize() + + graph_crore_multiplier = ( + graph_crore_single_thousand_hundred_remainder + | graph_digit_any + | graph_two_digit_any + | self.graph_exact_hundreds + | graph_hundred_remainder_any + | self.graph_exact_thousands + | self.graph_thousand_with_remainder + | self.graph_lakh + | self.graph_lakh_with_remainder + ).optimize() + + graph_crore_with_digit = ( + graph_crore_multiplier + + delete_space + + graph_crore_join_delete + + pynutil.insert("௦௦௦௦௦௦") + + delete_space + + graph_digit_any + ) + + graph_crore_with_two_digit = ( + graph_crore_multiplier + + delete_space + + graph_crore_join_delete + + pynutil.insert("௦௦௦௦௦") + + delete_space + + graph_two_digit_any + ) + + graph_crore_with_hundred = ( + graph_crore_multiplier + + delete_space + + graph_crore_join_delete + + pynutil.insert("௦௦௦௦") + + delete_space + + self.graph_exact_hundreds + ) + + graph_crore_with_hundred_remainder = ( + graph_crore_multiplier + + delete_space + + graph_crore_join_delete + + pynutil.insert("௦௦௦௦") + + delete_space + + graph_hundred_remainder_any + ) + + # Split by thousand-multiplier width so padding zeros are correct. + graph_crore_with_thousand_remainder = ( + graph_crore_multiplier + + delete_space + + graph_crore_join_delete + + pynutil.insert("௦௦௦") + + delete_space + + graph_thousand_with_remainder_single + ) | ( + graph_crore_multiplier + + delete_space + + graph_crore_join_delete + + pynutil.insert("௦௦") + + delete_space + + graph_thousand_with_remainder_two_digit + ) + + graph_crore_with_exact_single_lakh = ( + graph_crore_multiplier + + delete_space + + graph_crore_join_delete + + pynutil.insert("௦") + + delete_space + + graph_single_lakh + ) + + graph_crore_with_exact_two_digit_lakh = ( + graph_crore_multiplier + delete_space + graph_crore_join_delete + delete_space + graph_two_digit_lakh + ) + + graph_crore_with_exact_lakh = ( + graph_crore_with_exact_single_lakh | graph_crore_with_exact_two_digit_lakh + ).optimize() + + graph_single_lakh_with_digit = ( + graph_digit_any + + delete_space + + graph_lakh_join_delete + + pynutil.insert("௦௦௦௦") + + delete_space + + graph_digit_any + ) + + graph_single_lakh_with_two_digit = ( + graph_digit_any + + delete_space + + graph_lakh_join_delete + + pynutil.insert("௦௦௦") + + delete_space + + graph_two_digit_any + ) + + graph_single_lakh_with_hundred = ( + graph_digit_any + + delete_space + + graph_lakh_join_delete + + pynutil.insert("௦௦") + + delete_space + + self.graph_exact_hundreds + ) + + graph_single_lakh_with_hundred_remainder = ( + graph_digit_any + + delete_space + + graph_lakh_join_delete + + pynutil.insert("௦௦") + + delete_space + + graph_hundred_remainder_any + ) + + graph_single_lakh_with_thousand = graph_lakh_single_with_thousand + + graph_single_lakh_remainder = ( + graph_single_lakh_with_digit + | graph_single_lakh_with_two_digit + | graph_single_lakh_with_hundred + | graph_single_lakh_with_hundred_remainder + | graph_single_lakh_with_thousand + ).optimize() + + graph_two_digit_lakh_with_digit = ( + graph_two_digit_any + + delete_space + + graph_lakh_join_delete + + pynutil.insert("௦௦௦௦") + + delete_space + + graph_digit_any + ) + + graph_two_digit_lakh_with_two_digit = ( + graph_two_digit_any + + delete_space + + graph_lakh_join_delete + + pynutil.insert("௦௦௦") + + delete_space + + graph_two_digit_any + ) + + graph_two_digit_lakh_with_hundred = ( + graph_two_digit_any + + delete_space + + graph_lakh_join_delete + + pynutil.insert("௦௦") + + delete_space + + self.graph_exact_hundreds + ) + + graph_two_digit_lakh_with_hundred_remainder = ( + graph_two_digit_any + + delete_space + + graph_lakh_join_delete + + pynutil.insert("௦௦") + + delete_space + + graph_hundred_remainder_any + ) + + graph_two_digit_lakh_with_thousand = ( + graph_two_digit_any + + delete_space + + graph_lakh_join_delete + + delete_space + + self.graph_thousand_with_remainder + ) + + graph_two_digit_lakh_remainder = ( + graph_two_digit_lakh_with_digit + | graph_two_digit_lakh_with_two_digit + | graph_two_digit_lakh_with_hundred + | graph_two_digit_lakh_with_hundred_remainder + | graph_two_digit_lakh_with_thousand + ).optimize() + + graph_crore_with_single_lakh_remainder = ( + graph_crore_multiplier + + delete_space + + graph_crore_join_delete + + pynutil.insert("௦") + + delete_space + + graph_single_lakh_remainder + ) + + self.graph_crore_with_remainder = ( + graph_crore_with_digit + | graph_crore_with_two_digit + | graph_crore_with_hundred + | graph_crore_with_hundred_remainder + | graph_crore_with_thousand_remainder + | graph_crore_with_exact_lakh + | graph_crore_with_single_lakh_remainder + | ( + graph_crore_multiplier + + delete_space + + graph_crore_join_delete + + delete_space + + graph_two_digit_lakh_remainder + ) + ).optimize() + + graph = ( + graph_zero + | graph_digit + | self.graph_two_digit + | self.graph_exact_hundreds + | self.graph_hundred_with_remainder + | self.graph_exact_thousands + | self.graph_thousand_with_remainder + | self.graph_lakh + | self.graph_lakh_with_remainder + | self.graph_crore + | self.graph_crore_with_remainder + ).optimize() + + number_graph = pynutil.insert('integer: "') + graph + pynutil.insert('"') + final_graph = self.add_tokens(number_graph) + self.fst = final_graph.optimize() diff --git a/nemo_text_processing/inverse_text_normalization/ta/taggers/punctuation.py b/nemo_text_processing/inverse_text_normalization/ta/taggers/punctuation.py new file mode 100644 index 000000000..6c2c5277e --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/ta/taggers/punctuation.py @@ -0,0 +1,31 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# Copyright 2024 and onwards Google, Inc. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.inverse_text_normalization.ta.graph_utils import GraphFst + + +class PunctuationFst(GraphFst): + def __init__(self): + super().__init__(name="punctuation", kind="classify") + + s = "!#$%&\'()*+,-./:;<=>?@^_`{|}~" + punct = pynini.union(*s) + + graph = pynutil.insert('name: "') + punct + pynutil.insert('"') + + self.fst = graph.optimize() diff --git a/nemo_text_processing/inverse_text_normalization/ta/taggers/tokenize_and_classify.py b/nemo_text_processing/inverse_text_normalization/ta/taggers/tokenize_and_classify.py new file mode 100644 index 000000000..173afd071 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/ta/taggers/tokenize_and_classify.py @@ -0,0 +1,85 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import logging +import os + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.inverse_text_normalization.ta.graph_utils import ( + GraphFst, + delete_extra_space, + delete_space, + generator_main, +) +from nemo_text_processing.inverse_text_normalization.ta.taggers.cardinal import CardinalFst +from nemo_text_processing.inverse_text_normalization.ta.taggers.punctuation import PunctuationFst +from nemo_text_processing.inverse_text_normalization.ta.taggers.word import WordFst + + +class ClassifyFst(GraphFst): + """ + Tamil ITN tokenizer/classifier. + Supports Cardinal, Ordinal, Decimal, Word, and Punctuation. + """ + + def __init__( + self, + cache_dir: str = None, + whitelist: str = None, + overwrite_cache: bool = False, + input_case: str = "cased", + ): + super().__init__(name="tokenize_and_classify", kind="classify") + + far_file = None + if cache_dir is not None and cache_dir != "None": + os.makedirs(cache_dir, exist_ok=True) + far_file = os.path.join(cache_dir, "ta_itn.far") + + if not overwrite_cache and far_file and os.path.exists(far_file): + self.fst = pynini.Far(far_file, mode="r")["tokenize_and_classify"] + logging.info(f"ClassifyFst.fst restored from {far_file}") + else: + logging.info("Creating Tamil ITN grammars") + + cardinal = CardinalFst() + cardinal_graph = cardinal.fst + + word_graph = WordFst().fst + + punct_graph = PunctuationFst().fst + + classify = pynutil.add_weight(cardinal_graph, 1.1) | pynutil.add_weight(word_graph, 100) + + punct = pynutil.insert("tokens { ") + pynutil.add_weight(punct_graph, weight=1.1) + pynutil.insert(" }") + + token = pynutil.insert("tokens { ") + classify + pynutil.insert(" }") + + token_plus_punct = ( + pynini.closure(punct + pynutil.insert(" ")) + token + pynini.closure(pynutil.insert(" ") + punct) + ) + + graph = token_plus_punct + pynini.closure(delete_extra_space + token_plus_punct) + + graph = delete_space + graph + delete_space + + self.fst = graph.optimize() + + if far_file: + generator_main( + far_file, + {"tokenize_and_classify": self.fst}, + ) diff --git a/nemo_text_processing/inverse_text_normalization/ta/taggers/word.py b/nemo_text_processing/inverse_text_normalization/ta/taggers/word.py new file mode 100644 index 000000000..a4bf0b5ce --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/ta/taggers/word.py @@ -0,0 +1,27 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# Copyright 2024 and onwards Google, Inc. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.inverse_text_normalization.ta.graph_utils import NEMO_NOT_SPACE, GraphFst + + +class WordFst(GraphFst): + + def __init__(self): + super().__init__(name="word", kind="classify") + word = pynutil.insert("name: \"") + pynini.closure(NEMO_NOT_SPACE, 1) + pynutil.insert("\"") + self.fst = word.optimize() diff --git a/nemo_text_processing/inverse_text_normalization/ta/utils.py b/nemo_text_processing/inverse_text_normalization/ta/utils.py new file mode 100644 index 000000000..9a0bc0335 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/ta/utils.py @@ -0,0 +1,47 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import csv +import logging +import os + + +def get_abs_path(rel_path): + """ + Get absolute path + + Args: + rel_path: relative path to this file + + Returns absolute path + """ + abs_path = os.path.dirname(os.path.abspath(__file__)) + os.sep + rel_path + + if not os.path.exists(abs_path): + logging.warning(f'{abs_path} does not exist') + return abs_path + + +def load_labels(abs_path): + """ + loads relative path file as dictionary + + Args: + abs_path: absolute path + + Returns dictionary of mappings + """ + label_tsv = open(abs_path, encoding="utf-8") + labels = list(csv.reader(label_tsv, delimiter="\t")) + return labels diff --git a/nemo_text_processing/inverse_text_normalization/ta/verbalizers/__init__.py b/nemo_text_processing/inverse_text_normalization/ta/verbalizers/__init__.py new file mode 100644 index 000000000..63a296dae --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/ta/verbalizers/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License.import pynini diff --git a/nemo_text_processing/inverse_text_normalization/ta/verbalizers/cardinal.py b/nemo_text_processing/inverse_text_normalization/ta/verbalizers/cardinal.py new file mode 100644 index 000000000..ada9e6f27 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/ta/verbalizers/cardinal.py @@ -0,0 +1,54 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.inverse_text_normalization.ta.graph_utils import NEMO_NOT_QUOTE, GraphFst, delete_space + + +class CardinalFst(GraphFst): + """ + Finite state transducer for verbalizing cardinal + e.g. cardinal { integer: "௨௩" } -> ௨௩ + """ + + def __init__(self): + super().__init__(name="cardinal", kind="verbalize") + + optional_sign = pynini.closure( + pynutil.delete("negative:") + + delete_space + + pynutil.delete("\"") + + NEMO_NOT_QUOTE + + pynutil.delete("\"") + + delete_space, + 0, + 1, + ) + + graph = ( + pynutil.delete("integer:") + + delete_space + + pynutil.delete("\"") + + pynini.closure(NEMO_NOT_QUOTE, 1) + + pynutil.delete("\"") + ) + + self.numbers = graph + + graph = optional_sign + graph + + delete_tokens = self.delete_tokens(graph) + self.fst = delete_tokens.optimize() diff --git a/nemo_text_processing/inverse_text_normalization/ta/verbalizers/verbalize.py b/nemo_text_processing/inverse_text_normalization/ta/verbalizers/verbalize.py new file mode 100644 index 000000000..6d1b685c7 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/ta/verbalizers/verbalize.py @@ -0,0 +1,26 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License.import pynini +from nemo_text_processing.inverse_text_normalization.ta.graph_utils import GraphFst +from nemo_text_processing.inverse_text_normalization.ta.verbalizers.cardinal import CardinalFst +from nemo_text_processing.inverse_text_normalization.ta.verbalizers.word import WordFst + + +class VerbalizeFst(GraphFst): + def __init__(self): + super().__init__(name="verbalize", kind="verbalize") + + cardinal = CardinalFst() + word = WordFst() + + self.fst = (cardinal.fst | word.fst).optimize() diff --git a/nemo_text_processing/inverse_text_normalization/ta/verbalizers/verbalize_final.py b/nemo_text_processing/inverse_text_normalization/ta/verbalizers/verbalize_final.py new file mode 100644 index 000000000..bb413da96 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/ta/verbalizers/verbalize_final.py @@ -0,0 +1,46 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# Copyright 2024 and onwards Google, Inc. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.inverse_text_normalization.ta.graph_utils import GraphFst, delete_extra_space, delete_space +from nemo_text_processing.inverse_text_normalization.ta.verbalizers.verbalize import VerbalizeFst + + +class VerbalizeFinalFst(GraphFst): + """ + Tamil final verbalizer. + Currently supports only Cardinal. + """ + + def __init__(self): + super().__init__(name="verbalize_final", kind="verbalize") + + verbalize = VerbalizeFst().fst + + graph = ( + pynutil.delete("tokens") + + delete_space + + pynutil.delete("{") + + delete_space + + verbalize + + delete_space + + pynutil.delete("}") + ) + + graph = delete_space + pynini.closure(graph + delete_extra_space) + graph + delete_space + + self.fst = graph.optimize() diff --git a/nemo_text_processing/inverse_text_normalization/ta/verbalizers/word.py b/nemo_text_processing/inverse_text_normalization/ta/verbalizers/word.py new file mode 100644 index 000000000..cd0a1e5b1 --- /dev/null +++ b/nemo_text_processing/inverse_text_normalization/ta/verbalizers/word.py @@ -0,0 +1,26 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License.import pynini +import pynini +from pynini.lib import pynutil + +from nemo_text_processing.inverse_text_normalization.ta.graph_utils import NEMO_NOT_QUOTE, GraphFst + + +class WordFst(GraphFst): + def __init__(self): + super().__init__(name="word", kind="verbalize") + + graph = pynutil.delete('name: "') + pynini.closure(NEMO_NOT_QUOTE, 1) + pynutil.delete('"') + + self.fst = graph.optimize() diff --git a/tests/nemo_text_processing/ta/__init__.py b/tests/nemo_text_processing/ta/__init__.py new file mode 100644 index 000000000..63a296dae --- /dev/null +++ b/tests/nemo_text_processing/ta/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License.import pynini diff --git a/tests/nemo_text_processing/ta/data_inverse_text_normalization/test_cases_cardinal.txt b/tests/nemo_text_processing/ta/data_inverse_text_normalization/test_cases_cardinal.txt new file mode 100644 index 000000000..6536c6b47 --- /dev/null +++ b/tests/nemo_text_processing/ta/data_inverse_text_normalization/test_cases_cardinal.txt @@ -0,0 +1,54 @@ +நான்கு ஃபோர்கள்~௪ ஃபோர்கள் +ஆறு வீரர்கள் அவுட்~௬ வீரர்கள் அவுட் +ஒன்பிளஸ் எட்டு ப்ரோ~ஒன்பிளஸ் ௮ ப்ரோ +ஐந்து சார்ஜர்கள்~௫ சார்ஜர்கள் +நான்கு ஓவரில் பதினேழு ரன்கள்~௪ ஓவரில் ௧௭ ரன்கள் +ஐந்து சாக்லெட்டுகள் ஒன்பது டாஃபிகள்~௫ சாக்லெட்டுகள் ௯ டாஃபிகள் +பத்தாயிரத்து தொண்ணூற்று ஒன்பது~௧௦௦௯௯ +ஒரு இலட்சத்து ஒன்று~௧௦௦௦௦௧ +நூறு~௧௦௦ +முன்னூற்று ஒன்பது~௩௦௯ +எழுநூற்று தொண்ணூற்று எட்டு~௭௯௮ +ஐந்தாயிரம்~௫௦௦௦ +எட்டாயிரத்து நான்கு~௮௦௦௪ +ஒன்பதாயிரத்து பதினாறு~௯௦௧௬ +ஆயிரத்து தொள்ளாயிரத்து பன்னிரண்டு~௧௯௧௨ +இரண்டாயிரத்து இருநூற்று இருபத்து இரண்டு~௨௨௨௨ +பதினான்காயிரம்~௧௪௦௦௦ +பதினெட்டாயிரத்து ஆறு~௧௮௦௦௬ +இருபத்தாறாயிரத்து இருபத்து ஒன்று~௨௬௦௨௧ +தொண்ணூற்றாறாயிரத்து எண்ணூற்று பதினொன்று~௯௬௮௧௧ +நான்கு இலட்சம்~௪௦௦௦௦௦ +இரண்டு இலட்சத்து இரண்டு~௨௦௦௦௦௨ +ஏழு இலட்சத்து இருபது~௭௦௦௦௨௦ +ஒன்பது இலட்சத்து முன்னூற்று இருபத்து ஒன்று~௯௦௦௩௨௧ +எட்டு இலட்சத்து ஐந்தாயிரத்து முன்னூற்று இருபத்து ஒன்று~௮௦௫௩௨௧ +இருபத்து மூன்று இலட்சம்~௨௩௦௦௦௦௦ +பதினைந்து இலட்சத்து ஒன்று~௧௫௦௦௦௦௧ +இருபத்து ஏழு இலட்சத்து எண்ணூற்று இருபது~௨௭௦௦௮௨௦ +தொண்ணூற்று ஒரு இலட்சத்து முப்பத்தொன்றாயிரத்து எண்ணூற்று இருபத்து ஒன்பது~௯௧௩௧௮௨௯ +மூன்று கோடி~௩௦௦௦௦௦௦௦ +ஒரு கோடியே ஒன்று~௧௦௦௦௦௦௦௧ +ஏழு கோடியே பதின்மூன்று~௭௦௦௦௦௦௧௩ +நான்கு கோடியே தொள்ளாயிரத்து பதினொன்று~௪௦௦௦௦௯௧௧ +ஆறு கோடியே ஐந்தாயிரத்து தொள்ளாயிரத்து பதினொன்று~௬௦௦௦௫௯௧௧ +ஆறு கோடியே இருபத்தைந்தாயிரத்து தொள்ளாயிரத்து பதினொன்று~௬௦௦௨௫௯௧௧ +மூன்று கோடியே ஒரு இலட்சத்து இருபத்தைந்தாயிரத்து தொள்ளாயிரத்து பதினொன்று~௩௦௧௨௫௯௧௧ +இரண்டு கோடியே பதினேழு இலட்சத்து இருபத்தைந்தாயிரத்து தொள்ளாயிரத்து பதினொன்று~௨௧௭௨௫௯௧௧ +முப்பது கோடி~௩௦௦௦௦௦௦௦௦ +தொண்ணூற்று எட்டு இலட்சத்து எழுபத்தாறாயிரத்து எழுநூற்று எண்பத்து ஒன்பது~௯௮௭௬௭௮௯ +இருபத்து மூன்று இலட்சத்து நாற்பத்தைந்தாயிரத்து ஐநூற்று அறுபத்து ஏழு~௨௩௪௫௫௬௭ +ஒரு கோடியே இருபத்து மூன்று இலட்சத்து நாற்பத்தைந்தாயிரத்து ஐநூற்று அறுபத்து ஏழு~௧௨௩௪௫௫௬௭ +ஒரு கோடியே இருபத்து ஒன்று இலட்சத்து இருபத்து ஒன்றாயிரத்து இருநூற்று பன்னிரண்டு~௧௨௧௨௧௨௧௨ +நூற்று பன்னிரண்டு கோடியே இருபத்து மூன்று இலட்சத்து நாற்பத்தைந்தாயிரத்து ஐநூற்று அறுபத்து ஏழு~௧௧௨௨௩௪௫௫௬௭ +நூற்று இரண்டு கோடியே இருபத்து மூன்று இலட்சத்து நாற்பத்தைந்தாயிரத்து ஐநூற்று அறுபத்து ஏழு~௧௦௨௨௩௪௫௫௬௭ +ஆயிரத்து நூற்று இரண்டு கோடியே இருபத்து மூன்று இலட்சத்து நாற்பத்தைந்தாயிரத்து ஐநூற்று அறுபத்து ஏழு~௧௧௦௨௨௩௪௫௫௬௭ +ஐந்தாயிரத்து நூற்று இரண்டு கோடியே இருபத்து மூன்று இலட்சத்து நாற்பத்தைந்தாயிரத்து ஐநூற்று அறுபத்து ஏழு~௫௧௦௨௨௩௪௫௫௬௭ +எழுநூற்று இருபத்தைந்து~௭௨௫ +எழுநூற்று ஐம்பது~௭௫௦ +ஏழாயிரத்து ஐநூறு~௭௫௦௦ +ஏழாயிரத்து இருநூற்று ஐம்பது~௭௨௫௦ +நூற்று ஐம்பது~௧௫௦ +இருநூற்று ஐம்பது~௨௫௦ +ஆயிரத்து அறுநூற்று ஐம்பது~௧௬௫௦ +ஆயிரத்து அறுநூற்று இருபத்தைந்து~௧௬௨௫ \ No newline at end of file diff --git a/tests/nemo_text_processing/ta/test_cardinal.py b/tests/nemo_text_processing/ta/test_cardinal.py new file mode 100644 index 000000000..63ac11dba --- /dev/null +++ b/tests/nemo_text_processing/ta/test_cardinal.py @@ -0,0 +1,36 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License.import pynini +import pytest +from parameterized import parameterized + +from nemo_text_processing.inverse_text_normalization.inverse_normalize import InverseNormalizer + +from ..utils import CACHE_DIR, parse_test_case_file + + +class TestCardinal: + inverse_normalizer = InverseNormalizer( + lang="ta", + cache_dir=CACHE_DIR, + overwrite_cache=False, + ) + + @parameterized.expand(parse_test_case_file("ta/data_inverse_text_normalization/test_cases_cardinal.txt")) + @pytest.mark.unit + def test_denorm(self, test_input, expected): + pred = self.inverse_normalizer.inverse_normalize( + test_input, + verbose=False, + ) + assert pred == expected diff --git a/tests/nemo_text_processing/ta/test_sparrowhawk_inverse_text_normalization.sh b/tests/nemo_text_processing/ta/test_sparrowhawk_inverse_text_normalization.sh new file mode 100644 index 000000000..de2af470a --- /dev/null +++ b/tests/nemo_text_processing/ta/test_sparrowhawk_inverse_text_normalization.sh @@ -0,0 +1,29 @@ +#! /bin/sh + +PROJECT_DIR=/workspace/tests + +runtest () { + input=$1 + cd /workspace/sparrowhawk/documentation/grammars + + # read test file + while read testcase; do + IFS='~' read spoken written <<< $testcase + denorm_pred=$(echo $spoken | normalizer_main --config=sparrowhawk_configuration.ascii_proto 2>&1 | tail -n 1) + + # trim white space + written="$(echo -e "${written}" | sed -e 's/^[[:space:]]*//' -e 's/[[:space:]]*$//')" + denorm_pred="$(echo -e "${denorm_pred}" | sed -e 's/^[[:space:]]*//' -e 's/[[:space:]]*$//')" + + # input expected actual + assertEquals "$spoken" "$written" "$denorm_pred" + done < "$input" +} + +testITNCardinal() { + input=$PROJECT_DIR/ta/data_inverse_text_normalization/test_cases_cardinal.txt + runtest $input +} + +# Load shUnit2 +. $PROJECT_DIR/../shunit2/shunit2 diff --git a/tools/text_processing_deployment/pynini_export.py b/tools/text_processing_deployment/pynini_export.py index 73a4fc138..f9e292639 100644 --- a/tools/text_processing_deployment/pynini_export.py +++ b/tools/text_processing_deployment/pynini_export.py @@ -109,6 +109,7 @@ def parse_args(): 'ja', 'rw', 'ko', + 'ta', ], type=str, default='en', @@ -352,6 +353,13 @@ def parse_args(): ClassifyFst as TNClassifyFst, ) from nemo_text_processing.text_normalization.ko.verbalizers.verbalize import VerbalizeFst as TNVerbalizeFst + elif args.language == 'ta': + from nemo_text_processing.inverse_text_normalization.ta.taggers.tokenize_and_classify import ( + ClassifyFst as ITNClassifyFst, + ) + from nemo_text_processing.inverse_text_normalization.ta.verbalizers.verbalize import ( + VerbalizeFst as ITNVerbalizeFst, + ) else: raise KeyError(f"Language {args.language} is not defined for export.") output_dir = os.path.join(args.output_dir, f"{args.language}_{args.grammars}_{args.input_case}")