From 911f8c0a0264acda8d1df7840a46b0a328618871 Mon Sep 17 00:00:00 2001 From: michalharakal Date: Fri, 25 Sep 2026 15:22:36 +0200 Subject: [PATCH 1/3] moonshine stream: length cap and end-aligned final window Two fixes for short utterances, where the German streaming tiny model is documented as weakest. Cap the final decode budget at the model card's own limit, max_new_tokens = samples * 6.5/16000 + 2, instead of a fixed 24. A fixed budget lets a one-second clip keep decoding long after the audio is spent, which is when this model repeats itself ("Lauter, lauter"). The cap is floored at 4 so a one-word command still has room and ceilinged at the previous fixed budget, so nothing decodes longer than it did before. End-align the last encoder window. The encoder graph takes features only and has no attention mask, so a window that runs past the end of the audio has its tail zero-padded and the encoder attends to that silence as if it were speech. Pulling the start back so the window ends on the final frame makes every frame in it real audio. Only applied when the utterance is at least one window long and the pulled-back start does not run ahead of what is already finalized; a clip shorter than one window still needs a real mask. The finish trace line now reports the tokens used against the budget. --- .../native/moonshine_stream_jni.c | 38 +++++++++++++++---- 1 file changed, 31 insertions(+), 7 deletions(-) diff --git a/llm-runtime/iree-android/native/moonshine_stream_jni.c b/llm-runtime/iree-android/native/moonshine_stream_jni.c index 5c091bab..661a8a0f 100644 --- a/llm-runtime/iree-android/native/moonshine_stream_jni.c +++ b/llm-runtime/iree-android/native/moonshine_stream_jni.c @@ -83,8 +83,8 @@ #define FE_OUT (CHUNK + FE_LC) /* 68 frames out */ #define MAXMEM 256 /* compiled cross-memory pad (5.12 s of finalized speech) */ #define MAXTOK 48 -#define FINISH_TOKENS 24 /* C&C final budget: 5.12 s of speech never needs more; halves the - worst-case finalize when the decode loops instead of stopping */ +#define FINISH_TOKENS 24 /* hard ceiling for the final budget (see finish_budget) */ +#define TOKENS_PER_SAMPLE (6.5f / 16000.0f) /* the model card's cap: max_new_tokens = samples * 6.5/16000 + 2 */ #define HOP_TOKENS 6 /* max new tokens decoded per hop (budget guard) */ #define RESTART_HOPS 2 /* first hops re-decode exactly instead of incrementally */ #define PEEK_FRAMES 36 /* early-peek window: first partial at ~0.72 s instead of 1.28 s */ @@ -273,6 +273,19 @@ static iree_hal_buffer_view_t* run1(Engine* e, iree_runtime_session_t* s, const return out; } +/* The model card's output-length cap, in tokens, for `npcm` samples of audio: + * max_new_tokens = samples * 6.5 / 16000 + 2 + * A fixed budget lets a short clip keep decoding long after the audio is spent, which is exactly + * when this model loops ("Lauter, lauter", "Kanal, vorheriges Kanal, vorher") — the card names + * short clips as its weak point and capping the length as the cure. Floored at 4 so a one-word + * command still has room, ceilinged at the previous fixed budget so nothing decodes longer than before. */ +static int finish_budget(int npcm) { + int b = (int)(npcm * TOKENS_PER_SAMPLE) + 2; + if (b < 4) b = 4; + if (b > FINISH_TOKENS) b = FINISH_TOKENS; + return b; +} + /* Process one encoder window starting at feature frame `start`; finalize rows into e->mem. */ static int process_window(Engine* e, int start, int produced, int flush, int peek) { double t0 = now_ms(); @@ -697,16 +710,27 @@ JNIEXPORT jstring JNICALL JNIFN(nativeFinish)(JNIEnv* env, jobject thiz, jlong h Engine* e = (Engine*)(intptr_t)handle; if (!e) return NULL; double t0 = now_ms(); int produced = e->npcm / SPF; - /* flush remaining windows (zero-padded), lookahead released */ + /* Flush the remaining windows, lookahead released. The encoder graph takes features only — it has + * no attention mask — so a window running past the end of the audio has its tail zero-padded and + * the encoder attends to that silence as if it were speech. For the last window we therefore pull + * `start` back so the window ENDS on the final frame: every one of its CHUNK frames is then real + * audio, and the rows still missing from memory are computed with a full right context. Only + * possible when the utterance is at least one window long and the pulled-back start does not run + * ahead of what is already finalized; a clip shorter than CHUNK still needs a real mask. */ while (e->nmem < produced && e->nmem < MAXMEM) { - if (process_window(e, e->win * HOP, produced, 1, 0)) break; + int start = e->win * HOP; + if (start + CHUNK > produced && produced >= CHUNK) { + int aligned = produced - CHUNK; + if (aligned < start && aligned <= e->nmem) start = aligned; + } + if (process_window(e, start, produced, 1, 0)) break; e->win++; } jstring out = NULL; if (e->nmem > 0) { /* exact: fresh decode over the final memory */ release_self(e); e->ntoks = 0; e->pos = 0; e->text[0] = 0; - if (run_prefill(e, 1) == 0) run_steps(e, FINISH_TOKENS, 1); + if (run_prefill(e, 1) == 0) run_steps(e, finish_budget(e->npcm), 1); if (!e->ended_eos) drop_restart_suffix(e); /* C&C utterances are single sentences; on a silent tail the greedy decode restarts the * utterance instead of emitting EOS ("Go to the next. Go to next") — same failure and @@ -716,8 +740,8 @@ JNIEXPORT jstring JNICALL JNIFN(nativeFinish)(JNIEnv* env, jobject thiz, jlong h } const char* p = e->text; while (*p == ' ') ++p; out = (*env)->NewStringUTF(env, p); - TLOG("finish %.0fms nmem %d toks %d \"%s\"\n", - now_ms() - t0, e->nmem, e->ntoks, p); + TLOG("finish %.0fms nmem %d toks %d/%d \"%s\"\n", + now_ms() - t0, e->nmem, e->ntoks, finish_budget(e->npcm), p); } reset_utterance(e); return out; From 244fa9856267da9b03897d9b574b117a001d4bd8 Mon Sep 17 00:00:00 2001 From: michalharakal Date: Fri, 25 Sep 2026 15:50:16 +0200 Subject: [PATCH 2/3] moonshine stream: drop the end-aligned final window, keep the length cap Measurement separates the two changes, and only one of them earns its place. The decode budget can only bind below about 3.4 s of audio; above that the model card's formula exceeds the old fixed ceiling and the code is a no-op. That gives a control group. On 183 German command-and-control recordings, clips below the threshold finish 0.58 s sooner (paired median, faster in 135 of 172) and clips above it are unchanged (+0.09 s, 11 recordings). The cap does exactly what it can do and nothing where it cannot act, so the win is attributable rather than incidental. Median word error rate does not move. The end-aligned window has no measured effect at all. The 11 recordings where it acts alone are byte-identical between the two builds. It also recomputes rows already finalized in e->mem, since the pulled-back start can precede e->win * HOP and the guard permits that overlap, so it carries a correctness risk for no demonstrated benefit. Removed. Per-recording transcript differences between the builds were 17 of 187, all below the cap threshold, and all of them lose the start of the utterance, which a budget applied to a greedy decode cannot cause. They are more likely harness noise than either change, and this measurement does not claim them. --- .../native/moonshine_stream_jni.c | 26 +++++++------------ 1 file changed, 9 insertions(+), 17 deletions(-) diff --git a/llm-runtime/iree-android/native/moonshine_stream_jni.c b/llm-runtime/iree-android/native/moonshine_stream_jni.c index 661a8a0f..08c601a2 100644 --- a/llm-runtime/iree-android/native/moonshine_stream_jni.c +++ b/llm-runtime/iree-android/native/moonshine_stream_jni.c @@ -275,10 +275,13 @@ static iree_hal_buffer_view_t* run1(Engine* e, iree_runtime_session_t* s, const /* The model card's output-length cap, in tokens, for `npcm` samples of audio: * max_new_tokens = samples * 6.5 / 16000 + 2 - * A fixed budget lets a short clip keep decoding long after the audio is spent, which is exactly - * when this model loops ("Lauter, lauter", "Kanal, vorheriges Kanal, vorher") — the card names - * short clips as its weak point and capping the length as the cure. Floored at 4 so a one-word - * command still has room, ceilinged at the previous fixed budget so nothing decodes longer than before. */ + * A fixed budget lets a short clip keep decoding long after the audio is spent, and the greedy + * decode spends that budget restarting the utterance rather than stopping. Floored at 4 so a + * one-word command still has room, ceilinged at the previous fixed budget so nothing decodes + * longer than before — above ~3.4 s of audio the formula exceeds the ceiling and this is a no-op. + * Measured on 183 German command-and-control recordings on a Mali device: clips below that + * threshold finish 0.58 s sooner (paired median, faster in 135 of 172), clips above it are + * unchanged (+0.09 s, 11 recordings), and the median word error rate does not move. */ static int finish_budget(int npcm) { int b = (int)(npcm * TOKENS_PER_SAMPLE) + 2; if (b < 4) b = 4; @@ -710,20 +713,9 @@ JNIEXPORT jstring JNICALL JNIFN(nativeFinish)(JNIEnv* env, jobject thiz, jlong h Engine* e = (Engine*)(intptr_t)handle; if (!e) return NULL; double t0 = now_ms(); int produced = e->npcm / SPF; - /* Flush the remaining windows, lookahead released. The encoder graph takes features only — it has - * no attention mask — so a window running past the end of the audio has its tail zero-padded and - * the encoder attends to that silence as if it were speech. For the last window we therefore pull - * `start` back so the window ENDS on the final frame: every one of its CHUNK frames is then real - * audio, and the rows still missing from memory are computed with a full right context. Only - * possible when the utterance is at least one window long and the pulled-back start does not run - * ahead of what is already finalized; a clip shorter than CHUNK still needs a real mask. */ + /* flush remaining windows (zero-padded), lookahead released */ while (e->nmem < produced && e->nmem < MAXMEM) { - int start = e->win * HOP; - if (start + CHUNK > produced && produced >= CHUNK) { - int aligned = produced - CHUNK; - if (aligned < start && aligned <= e->nmem) start = aligned; - } - if (process_window(e, start, produced, 1, 0)) break; + if (process_window(e, e->win * HOP, produced, 1, 0)) break; e->win++; } jstring out = NULL; From 8756f1dacbbe791894e0a3b96eb247485fc84641 Mon Sep 17 00:00:00 2001 From: michalharakal Date: Fri, 25 Sep 2026 17:05:40 +0200 Subject: [PATCH 3/3] docs(changelog): the Moonshine final-decode budget under Unreleased Records the cap, the control-group measurement it rests on, and the fact that the end-aligned window was dropped for measuring nothing. #458 keeps the real encoder-mask gap open. --- CHANGELOG.md | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 39032ef2..4d21458d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -74,6 +74,24 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 prefill is itself out-of-distribution for it). Measured on an 8-utterance zero-shot tool-calling set: 0/8 and 1/8 under generic ChatML / `QwenChatTemplate` vs 4-5/8 with this template. +### Changed — Moonshine v2 streaming: the final decode budget follows the model card (#455) + +- **`nativeFinish` caps the exact re-decode at `max_new_tokens = samples * 6.5/16000 + 2`** + (`llm-runtime:iree-android`, `moonshine_stream_jni.c`), the output-length cap the Moonshine v2 + model card specifies, instead of a fixed 24 tokens regardless of how much audio the utterance + held. Without it a short clip keeps decoding long after the audio is spent, and the greedy decode + spends the remaining budget restarting the utterance rather than stopping. The budget is floored + at 4 so a one-word command still has room and ceilinged at the previous fixed 24, so nothing + decodes longer than before — above roughly 3.4 s of audio the formula exceeds the ceiling and the + change is a no-op. The `finish` trace line now reports tokens used against the budget. +- **Measured** on 183 German command-and-control recordings on a Mali device, using clips above the + threshold as a control group: below it the turn finishes **0.58 s sooner** (paired median, faster + in 135 of 172); above it unchanged (+0.09 s, 11 recordings). Median word error rate does not move. + The saving appears exactly where the cap can act and nowhere else. +- The zero-padded final encoder window is a separate, still-open gap (#458): the encoder graph takes + features only and carries no attention mask, and moving the window is not the fix — a variant that + end-aligned it measured no effect on those same 11 control recordings and was dropped from #455. + ## [0.56.2] — 2026-09-22 A transformers-only release against **SKaiNET engine 0.56.0** (unchanged).