From 222415efefc469454bedb8fb4dca6ecc3970b0ce Mon Sep 17 00:00:00 2001 From: salee Date: Sat, 26 Sep 2026 18:31:49 +0900 Subject: [PATCH] qwen3_asr: drop stale language-hint caveats Language hints have been accepted since ccc2db0 (encode_language_prefix, covered by qwen3_asr_e2e_smoke), but the model docs, HF card summaries, family note, and fixture comment still said any explicit hint returns TRANSCRIBE_ERR_UNSUPPORTED_LANGUAGE. Remove those passages. --- docs/models/qwen3-asr-0.6b.md | 12 +----------- docs/models/qwen3-asr-1.7b.md | 10 +--------- docs/porting/families/qwen3_asr.md | 8 -------- scripts/hf_cards/qwen3-asr-0.6b.yaml | 2 +- scripts/hf_cards/qwen3-asr-1.7b.yaml | 3 +-- tests/fixtures/make_gguf_fixtures.py | 6 +----- 6 files changed, 5 insertions(+), 36 deletions(-) diff --git a/docs/models/qwen3-asr-0.6b.md b/docs/models/qwen3-asr-0.6b.md index 93e2c4e3..86a0910c 100644 --- a/docs/models/qwen3-asr-0.6b.md +++ b/docs/models/qwen3-asr-0.6b.md @@ -7,7 +7,7 @@ Offline multilingual speech-to-text. An 18-layer bidirectional audio encoder feeds a 28-layer Qwen3 causal LM with audio-token injection (fused audio+text sequence, no cross-attention). Auto-detects the audio's language across 30 languages and emits the transcript in that language. Takes a -16 kHz mono WAV; explicit language hints are not supported at this time. +16 kHz mono WAV and produces a transcript. ## What it's for @@ -113,16 +113,6 @@ If your audio is not already 16 kHz mono WAV, convert it first: ffmpeg -i input.mp3 -ar 16000 -ac 1 output.wav ``` -## Public API caveat — language hints - -This port accepts `params.language == NULL` (auto-detect) and -**rejects any explicit language hint** with -`TRANSCRIBE_ERR_UNSUPPORTED_LANGUAGE`. The `capabilities.languages` -list documents the 30 languages the model can auto-detect, not a set -of caller-settable hints. Rendering caller-supplied hints into the -chat template is tracked as follow-up work; see the family note at -`docs/porting/families/qwen3_asr.md` for details. - ## Performance ### Apple M4 Max diff --git a/docs/models/qwen3-asr-1.7b.md b/docs/models/qwen3-asr-1.7b.md index c092028c..d2cc4ccb 100644 --- a/docs/models/qwen3-asr-1.7b.md +++ b/docs/models/qwen3-asr-1.7b.md @@ -8,8 +8,7 @@ Offline multilingual speech-to-text. Same audio-LLM architecture as the audio-token injection), wider: encoder `d_model=1024` (16 heads), LM `hidden_size=2048`, `intermediate_size=6144`. Auto-detects the audio's language across 30 languages and emits the transcript in that language. -Takes a 16 kHz mono WAV; explicit language hints are not supported at -this time. +Takes a 16 kHz mono WAV and produces a transcript. ## What it's for @@ -107,13 +106,6 @@ If your audio is not already 16 kHz mono WAV, convert it first: ffmpeg -i input.mp3 -ar 16000 -ac 1 output.wav ``` -## Public API caveat — language hints - -Same contract as the 0.6B: `params.language == NULL` runs auto-detect; -any explicit hint returns `TRANSCRIBE_ERR_UNSUPPORTED_LANGUAGE`. See the -0.6B doc and the family note at `docs/porting/families/qwen3_asr.md` for -the rationale and the planned follow-up. - ## Performance ### Apple M4 Max diff --git a/docs/porting/families/qwen3_asr.md b/docs/porting/families/qwen3_asr.md index f3bf8468..1cabda24 100644 --- a/docs/porting/families/qwen3_asr.md +++ b/docs/porting/families/qwen3_asr.md @@ -132,14 +132,6 @@ Bridge validation: Things the first port intentionally does not do; tracked as follow- up work rather than shipped-and-broken. -- **Language hinting is rejected.** `transcribe_run_params.language == NULL` - is the supported mode and triggers the model's built-in auto-detect - (it prefixes the transcript with `language X`, which we strip before - returning). Any non-null hint, including a language in the - capability list, returns `TRANSCRIBE_ERR_UNSUPPORTED_LANGUAGE`. The - `caps.languages` list documents the model's auto-detect coverage, - not what callers may hint — rendering caller-supplied hints into the - chat template is a future change. - **Streaming** (`stream_transcribe` / chunk rollback) is out of scope for this port. Upstream Qwen3-ASR may be architecturally usable in a streaming mode, but the transcribe.cpp library does not expose or diff --git a/scripts/hf_cards/qwen3-asr-0.6b.yaml b/scripts/hf_cards/qwen3-asr-0.6b.yaml index 573a9fd4..3a61da6c 100644 --- a/scripts/hf_cards/qwen3-asr-0.6b.yaml +++ b/scripts/hf_cards/qwen3-asr-0.6b.yaml @@ -25,7 +25,7 @@ summary: | feeds a 28-layer Qwen3 causal LM with audio-token injection (fused audio+text sequence, no cross-attention). Auto-detects the audio's language across 30 languages and emits the transcript in that language. Takes a - 16 kHz mono WAV; explicit language hints are not supported at this time. + 16 kHz mono WAV and produces a transcript. wer: notes: | diff --git a/scripts/hf_cards/qwen3-asr-1.7b.yaml b/scripts/hf_cards/qwen3-asr-1.7b.yaml index d37ae837..4b37c735 100644 --- a/scripts/hf_cards/qwen3-asr-1.7b.yaml +++ b/scripts/hf_cards/qwen3-asr-1.7b.yaml @@ -26,8 +26,7 @@ summary: | audio-token injection), wider: encoder `d_model=1024` (16 heads), LM `hidden_size=2048`, `intermediate_size=6144`. Auto-detects the audio's language across 30 languages and emits the transcript in that language. - Takes a 16 kHz mono WAV; explicit language hints are not supported at - this time. + Takes a 16 kHz mono WAV and produces a transcript. wer: notes: | diff --git a/tests/fixtures/make_gguf_fixtures.py b/tests/fixtures/make_gguf_fixtures.py index ba9a1014..bd378c46 100644 --- a/tests/fixtures/make_gguf_fixtures.py +++ b/tests/fixtures/make_gguf_fixtures.py @@ -1757,11 +1757,7 @@ def emit_fixtures(out_dir: Path) -> None: _pack_kv_string("stt.variant", "qwen3-asr-toy"), _pack_kv_string("tokenizer.chat_template", QWEN3_ASR_CHAT_TEMPLATE), - # Capabilities: match the family note. The loader reads - # general.languages as factual detection coverage; the - # run() handler rejects any explicit language hint with - # TRANSCRIBE_ERR_UNSUPPORTED_LANGUAGE until the prompt - # renderer honors them (Phase 1.1 Option A). + # Capabilities: match the family note. _pack_kv_bool("stt.capability.lang_detect", True), _pack_kv_array_string( "general.languages",