From 760ab1ef1d1d59589d659480ab83cbb92d5128da Mon Sep 17 00:00:00 2001 From: Alex-Wengg Date: Wed, 23 Sep 2026 10:02:08 -0400 Subject: [PATCH] perf(fst): borrow the grammar in apply instead of cloning it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit engine::apply composed `byte_acceptor(input)` with `fst.clone()`, which deep-copied the whole grammar FST on every call. apply runs once for classify and once per token (per field permutation) for verbalize, and es verbalize is the largest bundled grammar (1.6 MB gzipped). rustfst's compose takes any Borrow + Clone, so pass the reference. FluidAudio's Kokoro Spanish TTS frontend runs this on every utterance (FluidAudio #926/#950). FLEURS test transcripts, release, M5 Pro: es 115.8 → 5.2 ms/sentence (281), peak RSS 208 → 133 MB fr 12.1 → 7.3 ms/sentence (262), peak RSS 61 → 44 MB Output is byte-identical on both sets, and the fst_parity suite passes. --- src/fst/engine.rs | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/src/fst/engine.rs b/src/fst/engine.rs index e5bb028..212fdbf 100644 --- a/src/fst/engine.rs +++ b/src/fst/engine.rs @@ -122,8 +122,10 @@ fn shortest_output(fst: &VectorFst) -> Option { /// shortest path's output. Returns `None` if the input is not in the domain /// (empty composition) or the shortest path emits nothing. pub fn apply(fst: &VectorFst, input: &str) -> Option { + // Borrow the grammar: cloning it per call copied the whole FST (several MB + // for es verbalize) once per token. let mut composed: VectorFst = - compose(byte_acceptor(input), fst.clone()).ok()?; + compose::<_, VectorFst<_>, VectorFst<_>, _, _, _>(byte_acceptor(input), fst).ok()?; if composed.num_states() == 0 || composed.start().is_none() { return None; }