From 8bbb119356016d512498d5fadcf602cdd8ce0bb0 Mon Sep 17 00:00:00 2001 From: Alex-Wengg Date: Tue, 29 Sep 2026 22:13:56 -0400 Subject: [PATCH] feat(tn/en): opt-in roman-numeral list markers (`roman_enumerators`) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit FluidAudio #972: Kokoro read `(i)`, `(ii)`, `(iv)` as letters. NeMo's roman grammar is uppercase-only and keyword-anchored, so the compiled FST leaves every list-marker form untouched (verified: `(i) … (ii) … (iv)`, `(IV)`, `II. Scope`, `ii) noise` all pass through; only `World War II` converts). Add `NormalizeOptions::roman_enumerators` (default false, same shape as `disable_bare_second`) and `tn::en::roman::spell_enumerators`, a port of FluidAudio's `EnglishTextNormalizer` pre-pass: - `(ii)` anywhere unless glued to a letter (`f(x)`, `café(i)` stay) - `ii)` / `ii.` only at line start or after `; : , .` + whitespace; the dot form only before whitespace (`i.e.` stays) - `I V X` only, single case, strict form, 1–39 (keeps `mix`, `cd`, `xl` out) - markers that aren't self-evident list items (uppercase, lone `v`/`x`, all-`x`) need a second enumerator in the text, so `(IV)` intravenous, checkbox `(x)`, `xx.` sign-offs, `v. Madison`, `I. M. Pei` stand alone while an uppercase outline `I. … II. … III.` converts whole Plumbing: `tn_normalize_sentence_lang_with_options` (rules engine), `fst::normalize_lang{,_with_options}` (FST engine), FFI `nemo_tn_fst_with_options` + `nemo_tn_normalize_sentence_lang_with_options`, WASM `tnNormalizeSentenceLangWithOptions`, Swift wrapper overload, headers. With the flag off every path is byte-identical to before (FST parity suite and `nemo_tn_fst` unchanged; FFI test asserts off == plain). --- src/ffi.rs | 147 ++++++++-- src/fst/mod.rs | 27 ++ src/lib.rs | 42 +++ src/options.rs | 16 ++ src/tn/en/roman.rs | 267 ++++++++++++++++++ src/wasm.rs | 21 +- .../include/nemo_text_processing.h | 25 ++ swift/NemoTextProcessing.swift | 28 ++ .../nemo_text_processing.h | 25 ++ tests/en_tn_tests.rs | 40 +++ 10 files changed, 619 insertions(+), 19 deletions(-) diff --git a/src/ffi.rs b/src/ffi.rs index b35926e..ddcafee 100644 --- a/src/ffi.rs +++ b/src/ffi.rs @@ -6,8 +6,9 @@ use std::ptr; use crate::{ custom_rules, normalize, normalize_sentence, normalize_sentence_lang, normalize_sentence_with_options, normalize_with_options, tn_normalize, tn_normalize_lang, - tn_normalize_sentence, tn_normalize_sentence_lang, tn_normalize_sentence_with_max_span, - tn_normalize_sentence_with_max_span_lang, NormalizeOptions, + tn_normalize_sentence, tn_normalize_sentence_lang, tn_normalize_sentence_lang_with_options, + tn_normalize_sentence_with_max_span, tn_normalize_sentence_with_max_span_lang, + NormalizeOptions, }; /// Build [`NormalizeOptions`] from FFI primitives. @@ -24,6 +25,7 @@ fn options_from_ffi( concat_compound_numbers: u32, max_span_tokens: u32, disable_bare_second: u32, + roman_enumerators: u32, ) -> NormalizeOptions { NormalizeOptions { concat_compound_numbers: concat_compound_numbers != 0, @@ -33,6 +35,7 @@ fn options_from_ffi( Some(max_span_tokens as usize) }, disable_bare_second: disable_bare_second != 0, + roman_enumerators: roman_enumerators != 0, } } @@ -115,7 +118,7 @@ pub unsafe extern "C" fn nemo_normalize_with_options( Err(_) => return ptr::null_mut(), }; - let options = options_from_ffi(concat_compound_numbers, 0, disable_bare_second); + let options = options_from_ffi(concat_compound_numbers, 0, disable_bare_second, 0); let result = normalize_with_options(c_str, options); match CString::new(result) { @@ -159,6 +162,7 @@ pub unsafe extern "C" fn nemo_normalize_sentence_with_options( concat_compound_numbers, max_span_tokens, disable_bare_second, + 0, ); let result = normalize_sentence_with_options(c_str, options); @@ -485,7 +489,7 @@ pub unsafe extern "C" fn nemo_tn_fst(input: *const c_char, lang: *const c_char) Ok(s) => s, Err(_) => return ptr::null_mut(), }; - match fst_normalize(input_str, lang_str) { + match fst_normalize(input_str, lang_str, NormalizeOptions::new()) { Some(result) => match CString::new(result) { Ok(c_string) => c_string.into_raw(), Err(_) => ptr::null_mut(), @@ -494,23 +498,82 @@ pub unsafe extern "C" fn nemo_tn_fst(input: *const c_char, lang: *const c_char) } } +/// `nemo_tn_fst` with caller options. Only `roman_enumerators` applies to the +/// FST path: non-zero reads English roman-numeral list markers as numbers +/// (`(ii)` → `(two)`) before the grammars run; zero is byte-exact NeMo. +/// +/// # Safety +/// - `input` and `lang` must be valid null-terminated UTF-8 strings +/// - Returns a newly allocated string that must be freed with `nemo_free_string` +#[no_mangle] +pub unsafe extern "C" fn nemo_tn_fst_with_options( + input: *const c_char, + lang: *const c_char, + roman_enumerators: u32, +) -> *mut c_char { + if input.is_null() || lang.is_null() { + return ptr::null_mut(); + } + let input_str = match CStr::from_ptr(input).to_str() { + Ok(s) => s, + Err(_) => return ptr::null_mut(), + }; + let lang_str = match CStr::from_ptr(lang).to_str() { + Ok(s) => s, + Err(_) => return ptr::null_mut(), + }; + let options = options_from_ffi(0, 0, 0, roman_enumerators); + match fst_normalize(input_str, lang_str, options) { + Some(result) => match CString::new(result) { + Ok(c_string) => c_string.into_raw(), + Err(_) => ptr::null_mut(), + }, + None => ptr::null_mut(), + } +} + +/// Normalize a full sentence (TN) for a specific language with caller options. +/// +/// `max_span_tokens`: `0` for the library default (`16`). `roman_enumerators`: +/// non-zero reads English roman-numeral list markers as numbers (`(ii)` → +/// `(two)`) before the taggers run. +/// +/// # Safety +/// - `input` and `lang` must be valid null-terminated UTF-8 strings +/// - Returns a newly allocated string that must be freed with `nemo_free_string` +#[no_mangle] +pub unsafe extern "C" fn nemo_tn_normalize_sentence_lang_with_options( + input: *const c_char, + lang: *const c_char, + max_span_tokens: u32, + roman_enumerators: u32, +) -> *mut c_char { + if input.is_null() || lang.is_null() { + return ptr::null_mut(); + } + let input_str = match CStr::from_ptr(input).to_str() { + Ok(s) => s, + Err(_) => return ptr::null_mut(), + }; + let lang_str = match CStr::from_ptr(lang).to_str() { + Ok(s) => s, + Err(_) => return ptr::null_mut(), + }; + let options = options_from_ffi(0, max_span_tokens, 0, roman_enumerators); + let result = tn_normalize_sentence_lang_with_options(input_str, lang_str, options); + match CString::new(result) { + Ok(c_string) => c_string.into_raw(), + Err(_) => ptr::null_mut(), + } +} + #[cfg(feature = "fst-engine")] -fn fst_normalize(input: &str, lang: &str) -> Option { - use crate::fst; - Some(match lang { - "en" => fst::en::normalize(input), - "zh" => fst::zh::normalize(input), - "ja" => fst::ja::normalize(input), - "fr" => fst::fr::normalize(input), - "es" => fst::es::normalize(input), - "de" => fst::de::normalize(input), - "hi" => fst::hi::normalize(input), - _ => return None, - }) +fn fst_normalize(input: &str, lang: &str, options: NormalizeOptions) -> Option { + crate::fst::normalize_lang_with_options(input, lang, options) } #[cfg(not(feature = "fst-engine"))] -fn fst_normalize(_input: &str, _lang: &str) -> Option { +fn fst_normalize(_input: &str, _lang: &str, _options: NormalizeOptions) -> Option { None } @@ -555,6 +618,56 @@ mod tests { } } + #[test] + fn test_ffi_tn_sentence_lang_with_options_roman_enumerators() { + unsafe { + let input = CString::new("(i) pay $5; (ii) leave").unwrap(); + let en = CString::new("en").unwrap(); + let off = + nemo_tn_normalize_sentence_lang_with_options(input.as_ptr(), en.as_ptr(), 0, 0); + assert_eq!( + CStr::from_ptr(off).to_str().unwrap(), + "(i) pay five dollars; (ii) leave" + ); + nemo_free_string(off); + let on = + nemo_tn_normalize_sentence_lang_with_options(input.as_ptr(), en.as_ptr(), 0, 1); + assert_eq!( + CStr::from_ptr(on).to_str().unwrap(), + "(one) pay five dollars; (two) leave" + ); + nemo_free_string(on); + } + } + + #[cfg(feature = "fst-engine")] + #[test] + fn test_ffi_tn_fst_with_options_roman_enumerators() { + unsafe { + let input = CString::new("(i) pay $5; (ii) leave").unwrap(); + let en = CString::new("en").unwrap(); + // Flag off is byte-identical to nemo_tn_fst (NeMo parity). + let plain = nemo_tn_fst(input.as_ptr(), en.as_ptr()); + let off = nemo_tn_fst_with_options(input.as_ptr(), en.as_ptr(), 0); + assert_eq!( + CStr::from_ptr(plain).to_str().unwrap(), + CStr::from_ptr(off).to_str().unwrap() + ); + assert_eq!( + CStr::from_ptr(off).to_str().unwrap(), + "(i) pay five dollars; (ii) leave" + ); + nemo_free_string(plain); + nemo_free_string(off); + let on = nemo_tn_fst_with_options(input.as_ptr(), en.as_ptr(), 1); + assert_eq!( + CStr::from_ptr(on).to_str().unwrap(), + "(one) pay five dollars; (two) leave" + ); + nemo_free_string(on); + } + } + #[test] fn test_ffi_normalize_with_options_concat_compound() { unsafe { diff --git a/src/fst/mod.rs b/src/fst/mod.rs index 3a4881f..8cab531 100644 --- a/src/fst/mod.rs +++ b/src/fst/mod.rs @@ -31,10 +31,37 @@ pub mod hi; pub mod ja; pub mod zh; +use crate::NormalizeOptions; use flate2::read::GzDecoder; use rustfst::prelude::*; use std::io::Read; +/// Normalize `input` with the grammars for `lang` (`en`, `zh`, `ja`, `fr`, +/// `es`, `de`, `hi`); `None` for an unsupported code. +pub fn normalize_lang(input: &str, lang: &str) -> Option { + Some(match lang { + "en" => en::normalize(input), + "zh" => zh::normalize(input), + "ja" => ja::normalize(input), + "fr" => fr::normalize(input), + "es" => es::normalize(input), + "de" => de::normalize(input), + "hi" => hi::normalize(input), + _ => return None, + }) +} + +/// [`normalize_lang`] with the option-gated pre-pass applied first. Only +/// [`NormalizeOptions::roman_enumerators`] affects this path; with every flag +/// off the output is byte-exact NeMo. +pub fn normalize_lang_with_options( + input: &str, + lang: &str, + options: NormalizeOptions, +) -> Option { + normalize_lang(&crate::tn_prepass(input, lang, options), lang) +} + /// Decompress a bundled `*.fst.gz` grammar and load it as an FST. fn load_gz(gz: &[u8]) -> VectorFst { let mut bytes = Vec::new(); diff --git a/src/lib.rs b/src/lib.rs index cd1d691..f28c8cc 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -1543,6 +1543,48 @@ pub fn tn_normalize_sentence_lang(input: &str, lang: &str) -> String { tn_normalize_sentence_with_max_span_lang(input, lang, DEFAULT_MAX_SPAN_TOKENS) } +/// Normalize a full sentence (TN) for a specific language with caller options. +/// +/// Honors [`NormalizeOptions::max_span_tokens`] and +/// [`NormalizeOptions::roman_enumerators`] (English only: `"(ii)"` → +/// `"(two)"` before the taggers run). The ITN-only flags are ignored. +/// +/// ``` +/// use text_processing_rs::{tn_normalize_sentence_lang_with_options, NormalizeOptions}; +/// +/// let opts = NormalizeOptions::new().with_roman_enumerators(true); +/// assert_eq!( +/// tn_normalize_sentence_lang_with_options("(i) pay $5; (ii) leave", "en", opts), +/// "(one) pay five dollars; (two) leave" +/// ); +/// ``` +pub fn tn_normalize_sentence_lang_with_options( + input: &str, + lang: &str, + options: NormalizeOptions, +) -> String { + let prepared = tn_prepass(input, lang, options); + tn_normalize_sentence_with_max_span_lang( + &prepared, + lang, + options.max_span_tokens.unwrap_or(DEFAULT_MAX_SPAN_TOKENS), + ) +} + +/// Option-gated rewrites that run before any TN engine (rule-based or FST): +/// currently only the English roman-numeral list-marker pass. +pub(crate) fn tn_prepass<'a>( + input: &'a str, + lang: &str, + options: NormalizeOptions, +) -> std::borrow::Cow<'a, str> { + if options.roman_enumerators && matches!(lang, "en" | "") { + std::borrow::Cow::Owned(tn::en::roman::spell_enumerators(input)) + } else { + std::borrow::Cow::Borrowed(input) + } +} + /// Normalize a full sentence (TN) for a specific language with configurable max span. pub fn tn_normalize_sentence_with_max_span_lang( input: &str, diff --git a/src/options.rs b/src/options.rs index fc9ab85..6e23196 100644 --- a/src/options.rs +++ b/src/options.rs @@ -25,6 +25,15 @@ pub struct NormalizeOptions { /// Compound ordinals (`"twenty second"` → `"22nd"`) and date contexts /// (`"January second twenty twenty five"`) still convert. Default `false`. pub disable_bare_second: bool, + + /// TN only (English): read roman-numeral list markers as numbers — + /// `"(ii)"` → `"(two)"`, `"ii)"` → `"two)"`, `"ii."` → `"two."` — before + /// the taggers run (FluidAudio #972). NeMo's roman grammar is + /// uppercase-only and keyword-anchored, so these otherwise pass through + /// and a TTS frontend reads them as letters. Off by default because it is + /// an extension beyond NeMo's output. See + /// [`crate::tn::en::roman::spell_enumerators`] for the exact rules. + pub roman_enumerators: bool, } impl NormalizeOptions { @@ -34,6 +43,7 @@ impl NormalizeOptions { concat_compound_numbers: false, max_span_tokens: None, disable_bare_second: false, + roman_enumerators: false, } } @@ -54,4 +64,10 @@ impl NormalizeOptions { self.disable_bare_second = enabled; self } + + /// Set [`Self::roman_enumerators`]. + pub const fn with_roman_enumerators(mut self, enabled: bool) -> Self { + self.roman_enumerators = enabled; + self + } } diff --git a/src/tn/en/roman.rs b/src/tn/en/roman.rs index 006ea43..60ee15f 100644 --- a/src/tn/en/roman.rs +++ b/src/tn/en/roman.rs @@ -57,6 +57,163 @@ fn is_name(word: &str) -> bool { && chars.all(|c| c.is_ascii_lowercase()) } +/// Rewrite roman-numeral list markers — `(ii)`, `ii)`, `ii.` — to spoken +/// cardinals, leaving the surrounding punctuation in place (`(two)`). Port of +/// FluidAudio's `EnglishTextNormalizer` pre-pass (FluidAudio #972); opt-in via +/// [`crate::NormalizeOptions::roman_enumerators`] because NeMo leaves these +/// markers untouched. +/// +/// Keyed on the enumerator *form*, never on the letters alone (`mix`, `did`, +/// `civil` and the pronoun `I` are all roman letters): +/// - `(ii)` may appear anywhere, as long as the `(` is not glued to a letter +/// (`f(x)` stays) and the `)` is not glued to a letter or digit. +/// - `ii)` / `ii.` only in enumerator position — line start (indentation +/// allowed) or after `; : , .` + whitespace — and the dot form only when +/// followed by whitespace (`i.e.` stays). +/// - Numerals use `I V X` only, in a single case, strict subtractive form, +/// 1–39. Admitting `L C D M` would put `mix`, `cd`, `mm`, `xl` in the trap set. +/// - A marker that is not self-evidently a list item — uppercase, a lone +/// `v`/`x`, or all-`x` (`xx`) — needs a second enumerator somewhere in the +/// text: medical `(IV)`, a checkbox `(x)`, a sign-off `xx.`, the citation +/// `v. Madison` and initials `I. M. Pei` stand alone, while an uppercase +/// outline `I. … II. … III.` converts as a whole. +pub fn spell_enumerators(text: &str) -> String { + let chars: Vec = text.chars().collect(); + let found = find_enumerators(&chars); + if found.is_empty() { + return text.to_string(); + } + let is_list = found.len() >= 2; + + let mut out = String::with_capacity(text.len()); + let mut last = 0; + for e in &found { + if !(is_list || is_self_evident(&e.numeral)) { + continue; + } + out.extend(&chars[last..e.start]); + out.push_str(&number_to_words(e.value)); + last = e.end; + } + out.extend(&chars[last..]); + out +} + +/// A well-formed roman numeral in an enumerator form. `start..end` covers the +/// numeral only; the surrounding punctuation is re-emitted verbatim. +struct Enumerator { + start: usize, + end: usize, + numeral: String, + value: i64, +} + +fn find_enumerators(chars: &[char]) -> Vec { + let mut found = Vec::new(); + let mut i = 0; + while i < chars.len() { + if !is_roman_letter(chars[i]) { + i += 1; + continue; + } + // Maximal run of roman letters starting here. + let mut end = i; + while end < chars.len() && is_roman_letter(chars[end]) { + end += 1; + } + let numeral: String = chars[i..end].iter().collect(); + let uniform_case = numeral.chars().all(|c| c.is_ascii_lowercase()) + || numeral.chars().all(|c| c.is_ascii_uppercase()); + if numeral.len() <= 7 && uniform_case && in_enumerator_form(chars, i, end) { + if let Some(value) = strict_value(&numeral) { + found.push(Enumerator { + start: i, + end, + numeral, + value, + }); + } + } + i = end; + } + found +} + +fn is_roman_letter(c: char) -> bool { + matches!(c, 'i' | 'v' | 'x' | 'I' | 'V' | 'X') +} + +/// `(ii)`, or `ii)` / `ii.` in enumerator position. +fn in_enumerator_form(chars: &[char], start: usize, end: usize) -> bool { + let at = |idx: usize| chars.get(idx).copied(); + let before = start.checked_sub(1).and_then(at); + let after = at(end); + + if before == Some('(') && after == Some(')') { + let glued_left = start + .checked_sub(2) + .and_then(at) + .is_some_and(|c| c.is_alphabetic()); + let glued_right = at(end + 1).is_some_and(|c| c.is_alphanumeric()); + return !glued_left && !glued_right; + } + + let closer_ok = match after { + Some(')') => !at(end + 1).is_some_and(|c| c.is_alphanumeric()), + Some('.') => at(end + 1).is_some_and(char::is_whitespace), + _ => false, + }; + closer_ok && in_enumerator_position(chars, start) +} + +/// Line start (after optional indentation) or clause punctuation + whitespace. +fn in_enumerator_position(chars: &[char], start: usize) -> bool { + let mut j = start; + while j > 0 && matches!(chars[j - 1], ' ' | '\t') { + j -= 1; + } + if j == 0 || matches!(chars[j - 1], '\n' | '\r') { + return true; + } + start >= 2 + && chars[start - 1].is_whitespace() + && matches!(chars[start - 2], ';' | ':' | ',' | '.') +} + +/// Lowercase `i`, or a lowercase multi-letter numeral that is not all `x`: +/// nothing else reads as a word or an abbreviation in enumerator position. +fn is_self_evident(numeral: &str) -> bool { + if numeral.chars().any(|c| c.is_ascii_uppercase()) { + return false; + } + numeral == "i" || (numeral.len() >= 2 && numeral.chars().any(|c| c != 'x')) +} + +/// Value of a strict-form `I V X` numeral (1–39), else `None`: tens `X{0,3}`, +/// then units `IX | IV | V?I{0,3}`. +fn strict_value(numeral: &str) -> Option { + let upper = numeral.to_ascii_uppercase(); + let tens = upper.chars().take_while(|&c| c == 'X').count(); + if tens > 3 { + return None; + } + let units = match &upper[tens..] { + "" => 0, + "I" => 1, + "II" => 2, + "III" => 3, + "IV" => 4, + "V" => 5, + "VI" => 6, + "VII" => 7, + "VIII" => 8, + "IX" => 9, + _ => return None, + }; + let value = 10 * tens as i64 + units; + (value > 0).then_some(value) +} + /// Convert a roman numeral to its value, or `None` if it is not a valid /// numeral (empty, or containing a non-roman letter). fn roman_to_int(s: &str) -> Option { @@ -114,4 +271,114 @@ mod tests { assert_eq!(roman_to_int("IIII"), Some(4)); // lax: additive form allowed assert_eq!(roman_to_int("hi"), None); } + + // spell_enumerators — mirrors FluidAudio's EnglishTextNormalizerTests (#972). + + #[test] + fn enumerators_parenthesized() { + assert_eq!( + spell_enumerators("(i) pay rent; (ii) keep the peace; (iii) insure; (iv) vacate."), + "(one) pay rent; (two) keep the peace; (three) insure; (four) vacate." + ); + assert_eq!( + spell_enumerators("(ix) and (xiv) and (xxxix)"), + "(nine) and (fourteen) and (thirty nine)" + ); + assert_eq!( + spell_enumerators("see (IV) and (XII)"), + "see (four) and (twelve)" + ); + assert_eq!( + spell_enumerators("under 2(a)(ii) above"), + "under 2(a)(two) above" + ); + } + + #[test] + fn enumerators_half_paren_and_dot() { + assert_eq!( + spell_enumerators("as follows: i) rent; ii) noise; iii) pets"), + "as follows: one) rent; two) noise; three) pets" + ); + assert_eq!( + spell_enumerators("i) first\nii) second\n iv) fourth"), + "one) first\ntwo) second\n four) fourth" + ); + assert_eq!( + spell_enumerators("I) first; II) second"), + "one) first; two) second" + ); + assert_eq!( + spell_enumerators("i. Introduction\nii. Methods\niv. Results"), + "one. Introduction\ntwo. Methods\nfour. Results" + ); + assert_eq!( + spell_enumerators("I. Intro\nII. Body\nIII. End"), + "one. Intro\ntwo. Body\nthree. End" + ); + } + + #[test] + fn enumerators_single_lowercase_marker_is_enough() { + assert_eq!(spell_enumerators("(i) pay rent"), "(one) pay rent"); + assert_eq!( + spell_enumerators("(iv) vacate on notice"), + "(four) vacate on notice" + ); + assert_eq!(spell_enumerators("vi. Appendix"), "six. Appendix"); + } + + #[test] + fn enumerators_ambiguous_markers_need_list_context() { + for s in [ + "morphine (IV) fluids", + "Mark (x) here", + "(v) to run", + "(I) think", + "Thanks. xx. Jane", + "Marbury\nv. Madison", + "Solve for the variable, x. Then", + ] { + assert_eq!(spell_enumerators(s), s, "{s:?} should be unchanged"); + } + assert_eq!( + spell_enumerators("(IV) fluids; (V) rest"), + "(four) fluids; (five) rest" + ); + assert_eq!( + spell_enumerators("(ix) foo; (x) bar"), + "(nine) foo; (ten) bar" + ); + } + + #[test] + fn enumerators_prose_and_non_forms_unchanged() { + for s in [ + "mix it, did I? civil and mild", + "(and so did I)", + "I. M. Pei designed it", + "I use vi. It rocks", + "i.e. the rest", + "the variable x) is free", + "f(x) and g(i)", + "café(i) test", + "(xl) size", + "(mix) (cd) (mm)", + "(iiii) (vv) (Iv) (ivi)", + "", + ] { + assert_eq!(spell_enumerators(s), s, "{s:?} should be unchanged"); + } + } + + #[test] + fn strict_values() { + assert_eq!(strict_value("i"), Some(1)); + assert_eq!(strict_value("XXXIX"), Some(39)); + assert_eq!(strict_value("iv"), Some(4)); + assert_eq!(strict_value("IIII"), None); + assert_eq!(strict_value("XXXX"), None); + assert_eq!(strict_value("VX"), None); + assert_eq!(strict_value(""), None); + } } diff --git a/src/wasm.rs b/src/wasm.rs index 7763abb..d9b23b7 100644 --- a/src/wasm.rs +++ b/src/wasm.rs @@ -6,8 +6,8 @@ use crate::{ custom_rules, normalize, normalize_sentence, normalize_sentence_lang, normalize_sentence_with_options, normalize_with_lang, normalize_with_options, tn_normalize, tn_normalize_lang, tn_normalize_sentence, tn_normalize_sentence_lang, - tn_normalize_sentence_with_max_span, tn_normalize_sentence_with_max_span_lang, - NormalizeOptions, + tn_normalize_sentence_lang_with_options, tn_normalize_sentence_with_max_span, + tn_normalize_sentence_with_max_span_lang, NormalizeOptions, }; /// Build [`NormalizeOptions`] from JS-friendly primitives. @@ -27,6 +27,7 @@ fn js_options( Some(max_span_tokens as usize) }, disable_bare_second, + roman_enumerators: false, } } @@ -128,6 +129,22 @@ pub fn tn_normalize_sentence_with_max_span_lang_js( tn_normalize_sentence_with_max_span_lang(input, lang, max_span_tokens as usize) } +/// Sentence TN with options. `maxSpanTokens == 0` means "use library +/// default" (16). `romanEnumerators=true` reads English roman-numeral list +/// markers as numbers (`(ii)` → `(two)`) before the taggers run +/// (FluidAudio #972). +#[wasm_bindgen(js_name = tnNormalizeSentenceLangWithOptions)] +pub fn tn_normalize_sentence_lang_with_options_js( + input: &str, + lang: &str, + max_span_tokens: u32, + roman_enumerators: bool, +) -> String { + let options = + js_options(false, max_span_tokens, false).with_roman_enumerators(roman_enumerators); + tn_normalize_sentence_lang_with_options(input, lang, options) +} + #[wasm_bindgen(js_name = addRule)] pub fn add_rule_js(spoken: &str, written: &str) { custom_rules::add_rule(spoken, written); diff --git a/swift-test/Sources/CNemoTextProcessing/include/nemo_text_processing.h b/swift-test/Sources/CNemoTextProcessing/include/nemo_text_processing.h index 3318749..e62217e 100644 --- a/swift-test/Sources/CNemoTextProcessing/include/nemo_text_processing.h +++ b/swift-test/Sources/CNemoTextProcessing/include/nemo_text_processing.h @@ -35,6 +35,31 @@ char* nemo_tn_normalize_sentence_with_max_span(const char* input, uint32_t max_s /* Byte-exact NeMo TN via the compiled-FST engine (NULL if unavailable) */ char* nemo_tn_fst(const char* input, const char* lang); +/** + * nemo_tn_fst with caller options. Only roman_enumerators applies to the FST + * path: non-zero reads English roman-numeral list markers as numbers + * ("(ii)" -> "(two)") before the grammars run; zero is byte-exact NeMo. + * + * @param input Null-terminated UTF-8 string + * @param lang Null-terminated language code + * @param roman_enumerators 0 = off (NeMo parity), non-zero = on + * @return Newly allocated string (free with nemo_free_string), or NULL. + */ +char* nemo_tn_fst_with_options(const char* input, const char* lang, uint32_t roman_enumerators); + +/** + * Text Normalization: normalize a full sentence for a specific language with + * caller options. + * + * @param input Null-terminated UTF-8 string + * @param lang Null-terminated language code (e.g. "en", "fr") + * @param max_span_tokens Maximum consecutive tokens per span; 0 = library default (16) + * @param roman_enumerators Non-zero reads English roman-numeral list markers as + * numbers ("(ii)" -> "(two)") before the taggers run. + * @return Newly allocated string, must be freed with nemo_free_string(). + */ +char* nemo_tn_normalize_sentence_lang_with_options(const char* input, const char* lang, uint32_t max_span_tokens, uint32_t roman_enumerators); + #ifdef __cplusplus } #endif diff --git a/swift/NemoTextProcessing.swift b/swift/NemoTextProcessing.swift index 1768010..da58540 100644 --- a/swift/NemoTextProcessing.swift +++ b/swift/NemoTextProcessing.swift @@ -271,6 +271,34 @@ public enum NemoTextProcessing { return String(cString: resultPtr) } + /// Normalize a full sentence for a specific language with options. + /// + /// - Parameters: + /// - input: Sentence containing written-form spans + /// - language: ISO 639-1 language code + /// - maxSpanTokens: Maximum consecutive tokens per span; `0` = library default (16) + /// - romanEnumerators: When true, English roman-numeral list markers are + /// read as numbers (`"(ii)"` → `"(two)"`) before the taggers run + /// (FluidAudio #972). Off keeps NeMo's behavior, which leaves them as letters. + /// - Returns: Sentence with written-form spans replaced with spoken form + public static func tnNormalizeSentence( + _ input: String, + language: String, + maxSpanTokens: UInt32, + romanEnumerators: Bool + ) -> String { + guard let inputC = input.cString(using: .utf8), + let langC = language.cString(using: .utf8) else { + return input + } + let romanFlag: UInt32 = romanEnumerators ? 1 : 0 + guard let resultPtr = nemo_tn_normalize_sentence_lang_with_options(inputC, langC, maxSpanTokens, romanFlag) else { + return input + } + defer { nemo_free_string(resultPtr) } + return String(cString: resultPtr) + } + // MARK: - Custom Rules /// Add a custom spoken→written normalization rule. diff --git a/swift/include/CNemoTextProcessing/nemo_text_processing.h b/swift/include/CNemoTextProcessing/nemo_text_processing.h index 496e39c..d934bdb 100644 --- a/swift/include/CNemoTextProcessing/nemo_text_processing.h +++ b/swift/include/CNemoTextProcessing/nemo_text_processing.h @@ -187,6 +187,31 @@ char* nemo_tn_normalize_sentence_with_max_span_lang(const char* input, const cha */ char* nemo_tn_fst(const char* input, const char* lang); +/** + * nemo_tn_fst with caller options. Only roman_enumerators applies to the FST + * path: non-zero reads English roman-numeral list markers as numbers + * ("(ii)" -> "(two)") before the grammars run; zero is byte-exact NeMo. + * + * @param input Null-terminated UTF-8 string + * @param lang Null-terminated language code + * @param roman_enumerators 0 = off (NeMo parity), non-zero = on + * @return Newly allocated string (free with nemo_free_string), or NULL. + */ +char* nemo_tn_fst_with_options(const char* input, const char* lang, uint32_t roman_enumerators); + +/** + * Text Normalization: normalize a full sentence for a specific language with + * caller options. + * + * @param input Null-terminated UTF-8 string + * @param lang Null-terminated language code (e.g. "en", "fr") + * @param max_span_tokens Maximum consecutive tokens per span; 0 = library default (16) + * @param roman_enumerators Non-zero reads English roman-numeral list markers as + * numbers ("(ii)" -> "(two)") before the taggers run. + * @return Newly allocated string, must be freed with nemo_free_string(). + */ +char* nemo_tn_normalize_sentence_lang_with_options(const char* input, const char* lang, uint32_t max_span_tokens, uint32_t roman_enumerators); + /** * Free a string allocated by nemo_normalize or nemo_normalize_sentence. * diff --git a/tests/en_tn_tests.rs b/tests/en_tn_tests.rs index 5e30da2..8aad481 100644 --- a/tests/en_tn_tests.rs +++ b/tests/en_tn_tests.rs @@ -329,3 +329,43 @@ fn test_tn_range() { results.failures.len() ); } + +// FluidAudio #972: roman-numeral list markers are an opt-in extension beyond +// NeMo (`roman_enumerators`); the default path must keep passing them through. + +#[test] +fn test_roman_enumerators_default_unchanged() { + use text_processing_rs::tn_normalize_sentence_lang; + assert_eq!( + tn_normalize_sentence_lang("(i) pay rent; (ii) leave", "en"), + "(i) pay rent; (ii) leave" + ); +} + +#[test] +fn test_roman_enumerators_opt_in() { + use text_processing_rs::{tn_normalize_sentence_lang_with_options, NormalizeOptions}; + let opts = NormalizeOptions::new().with_roman_enumerators(true); + assert_eq!( + tn_normalize_sentence_lang_with_options( + "The tenant shall: (i) pay $5; (ii) keep the peace; (iv) vacate on notice.", + "en", + opts + ), + "The tenant shall: (one) pay five dollars; (two) keep the peace; (four) vacate on notice." + ); + // Stand-alone ambiguous markers and prose are untouched even when opted in. + assert_eq!( + tn_normalize_sentence_lang_with_options("morphine (IV) fluids", "en", opts), + "morphine (IV) fluids" + ); + assert_eq!( + tn_normalize_sentence_lang_with_options("mix it, did I?", "en", opts), + "mix it, did I?" + ); + // English-only: other languages ignore the flag. + assert_eq!( + tn_normalize_sentence_lang_with_options("(ii) chats", "fr", opts), + "(ii) chats" + ); +}