diff --git a/CHANGELOG.md b/CHANGELOG.md index 39032ef2..4d21458d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -74,6 +74,24 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 prefill is itself out-of-distribution for it). Measured on an 8-utterance zero-shot tool-calling set: 0/8 and 1/8 under generic ChatML / `QwenChatTemplate` vs 4-5/8 with this template. +### Changed — Moonshine v2 streaming: the final decode budget follows the model card (#455) + +- **`nativeFinish` caps the exact re-decode at `max_new_tokens = samples * 6.5/16000 + 2`** + (`llm-runtime:iree-android`, `moonshine_stream_jni.c`), the output-length cap the Moonshine v2 + model card specifies, instead of a fixed 24 tokens regardless of how much audio the utterance + held. Without it a short clip keeps decoding long after the audio is spent, and the greedy decode + spends the remaining budget restarting the utterance rather than stopping. The budget is floored + at 4 so a one-word command still has room and ceilinged at the previous fixed 24, so nothing + decodes longer than before — above roughly 3.4 s of audio the formula exceeds the ceiling and the + change is a no-op. The `finish` trace line now reports tokens used against the budget. +- **Measured** on 183 German command-and-control recordings on a Mali device, using clips above the + threshold as a control group: below it the turn finishes **0.58 s sooner** (paired median, faster + in 135 of 172); above it unchanged (+0.09 s, 11 recordings). Median word error rate does not move. + The saving appears exactly where the cap can act and nowhere else. +- The zero-padded final encoder window is a separate, still-open gap (#458): the encoder graph takes + features only and carries no attention mask, and moving the window is not the fix — a variant that + end-aligned it measured no effect on those same 11 control recordings and was dropped from #455. + ## [0.56.2] — 2026-09-22 A transformers-only release against **SKaiNET engine 0.56.0** (unchanged). diff --git a/llm-runtime/iree-android/native/moonshine_stream_jni.c b/llm-runtime/iree-android/native/moonshine_stream_jni.c index 5c091bab..08c601a2 100644 --- a/llm-runtime/iree-android/native/moonshine_stream_jni.c +++ b/llm-runtime/iree-android/native/moonshine_stream_jni.c @@ -83,8 +83,8 @@ #define FE_OUT (CHUNK + FE_LC) /* 68 frames out */ #define MAXMEM 256 /* compiled cross-memory pad (5.12 s of finalized speech) */ #define MAXTOK 48 -#define FINISH_TOKENS 24 /* C&C final budget: 5.12 s of speech never needs more; halves the - worst-case finalize when the decode loops instead of stopping */ +#define FINISH_TOKENS 24 /* hard ceiling for the final budget (see finish_budget) */ +#define TOKENS_PER_SAMPLE (6.5f / 16000.0f) /* the model card's cap: max_new_tokens = samples * 6.5/16000 + 2 */ #define HOP_TOKENS 6 /* max new tokens decoded per hop (budget guard) */ #define RESTART_HOPS 2 /* first hops re-decode exactly instead of incrementally */ #define PEEK_FRAMES 36 /* early-peek window: first partial at ~0.72 s instead of 1.28 s */ @@ -273,6 +273,22 @@ static iree_hal_buffer_view_t* run1(Engine* e, iree_runtime_session_t* s, const return out; } +/* The model card's output-length cap, in tokens, for `npcm` samples of audio: + * max_new_tokens = samples * 6.5 / 16000 + 2 + * A fixed budget lets a short clip keep decoding long after the audio is spent, and the greedy + * decode spends that budget restarting the utterance rather than stopping. Floored at 4 so a + * one-word command still has room, ceilinged at the previous fixed budget so nothing decodes + * longer than before — above ~3.4 s of audio the formula exceeds the ceiling and this is a no-op. + * Measured on 183 German command-and-control recordings on a Mali device: clips below that + * threshold finish 0.58 s sooner (paired median, faster in 135 of 172), clips above it are + * unchanged (+0.09 s, 11 recordings), and the median word error rate does not move. */ +static int finish_budget(int npcm) { + int b = (int)(npcm * TOKENS_PER_SAMPLE) + 2; + if (b < 4) b = 4; + if (b > FINISH_TOKENS) b = FINISH_TOKENS; + return b; +} + /* Process one encoder window starting at feature frame `start`; finalize rows into e->mem. */ static int process_window(Engine* e, int start, int produced, int flush, int peek) { double t0 = now_ms(); @@ -706,7 +722,7 @@ JNIEXPORT jstring JNICALL JNIFN(nativeFinish)(JNIEnv* env, jobject thiz, jlong h if (e->nmem > 0) { /* exact: fresh decode over the final memory */ release_self(e); e->ntoks = 0; e->pos = 0; e->text[0] = 0; - if (run_prefill(e, 1) == 0) run_steps(e, FINISH_TOKENS, 1); + if (run_prefill(e, 1) == 0) run_steps(e, finish_budget(e->npcm), 1); if (!e->ended_eos) drop_restart_suffix(e); /* C&C utterances are single sentences; on a silent tail the greedy decode restarts the * utterance instead of emitting EOS ("Go to the next. Go to next") — same failure and @@ -716,8 +732,8 @@ JNIEXPORT jstring JNICALL JNIFN(nativeFinish)(JNIEnv* env, jobject thiz, jlong h } const char* p = e->text; while (*p == ' ') ++p; out = (*env)->NewStringUTF(env, p); - TLOG("finish %.0fms nmem %d toks %d \"%s\"\n", - now_ms() - t0, e->nmem, e->ntoks, p); + TLOG("finish %.0fms nmem %d toks %d/%d \"%s\"\n", + now_ms() - t0, e->nmem, e->ntoks, finish_budget(e->npcm), p); } reset_utterance(e); return out;