diff --git a/packages/commands/src/commands/speech/recognize.ts b/packages/commands/src/commands/speech/recognize.ts index 31a2963..f6910fc 100644 --- a/packages/commands/src/commands/speech/recognize.ts +++ b/packages/commands/src/commands/speech/recognize.ts @@ -14,6 +14,7 @@ import { speechRecognizePath, resolveAsrApi, buildAsrFlashRequest, + buildAsyncAsrLanguageFields, extractAsrFlashText, type AsrApiRoute, type AsrFlashFamily, @@ -42,7 +43,7 @@ const RECOGNIZE_FLAGS = { type: "string", valueHint: "", description: - "Language hint (e.g. zh, en, ja). Async & input-audio sync: language_hints; qwen3 sync: asr_options.language", + "Language hint (e.g. zh, en, ja). Classic async/input-audio: language_hints; qwen3-filetrans: language; qwen3 sync: asr_options.language", }, diarization: { type: "switch", description: "Enable automatic speaker diarization" }, speakerCount: { @@ -70,11 +71,18 @@ const RECOGNIZE_FLAGS = { } satisfies FlagsDef; type RecognizeFlags = ParsedFlags; -function assertSyncFlashFlagsAllowed(flags: RecognizeFlags, model: string): void { +function assertSyncFlashFlagsAllowed( + flags: RecognizeFlags, + model: string, + flashFamily: AsrFlashFamily, +): void { const unsupported: string[] = []; if (flags.diarization === true) unsupported.push("--diarization"); if (flags.speakerCount !== undefined) unsupported.push("--speaker-count"); - if (flags.vocabularyId !== undefined) unsupported.push("--vocabulary-id"); + // qwen3 sync Flash 不走 vocabulary_id;input-audio Flash(fun-asr-flash* / qwen-audio-*-asr-flash)官方支持 + if (flashFamily === "qwen3" && flags.vocabularyId !== undefined) { + unsupported.push("--vocabulary-id"); + } if (flags.channelId !== undefined) unsupported.push("--channel-id"); if (flags.async === true) unsupported.push("--async"); if (flags.pollInterval !== undefined) unsupported.push("--poll-interval"); @@ -133,7 +141,7 @@ export default defineCommand({ } if (route.kind === "sync-flash") { - assertSyncFlashFlagsAllowed(flags, model); + assertSyncFlashFlagsAllowed(flags, model, route.flashFamily!); if (rawUrls.length !== 1) { throw new BailianError( `Model "${model}" is a sync Flash ASR model and accepts exactly one --url (got ${rawUrls.length}).\n` + @@ -173,8 +181,11 @@ export default defineCommand({ } const channelId = flags.channelId; - const language = flags.language; const vocabularyId = flags.vocabularyId; + const languageFields = buildAsyncAsrLanguageFields( + route.asyncLanguageStyle ?? "language_hints", + flags.language, + ); const body: DashScopeASRRequest = { model, @@ -184,7 +195,7 @@ export default defineCommand({ : { file_urls: resolvedUrls }, parameters: { channel_id: channelId !== undefined ? [channelId] : [0], - language_hints: language ? [language] : undefined, + ...languageFields, diarization_enabled: diarization ? true : undefined, speaker_count: speakerCount, vocabulary_id: vocabularyId, @@ -221,6 +232,7 @@ async function handleSyncFlashMode( model, audioUrl, language: flags.language, + vocabularyId: flags.vocabularyId, flashFamily, }); diff --git a/packages/commands/tests/e2e/speech-recognize.e2e.test.ts b/packages/commands/tests/e2e/speech-recognize.e2e.test.ts index 8612cc9..7ecf828 100644 --- a/packages/commands/tests/e2e/speech-recognize.e2e.test.ts +++ b/packages/commands/tests/e2e/speech-recognize.e2e.test.ts @@ -32,7 +32,12 @@ describe("e2e: speech recognize", () => { path?: string; request?: { model?: string; - parameters?: { format?: string; language_hints?: string[] }; + parameters?: { + format?: string; + language_hints?: string[]; + language?: string; + vocabulary_id?: string; + }; input?: { file_url?: string; file_urls?: string[]; @@ -60,26 +65,33 @@ describe("e2e: speech recognize", () => { "https://dashscope.oss-cn-beijing.aliyuncs.com/samples/audio/paraformer/hello_world_female2.wav", "--language", "en", + "--vocabulary-id", + "vocab-e2e", ]); expect(body.mode).toBe("sync"); expect(body.path).toBe("/api/v1/services/aigc/multimodal-generation/generation"); expect(body.request?.model).toBe("qwen-audio-3.0-asr-flash"); expect(body.request?.parameters?.format).toBe("wav"); expect(body.request?.parameters?.language_hints).toEqual(["en"]); + expect(body.request?.parameters?.vocabulary_id).toBe("vocab-e2e"); expect(body.request?.input?.messages?.[0]?.content?.[0]?.type).toBe("input_audio"); }); - test("speech recognize qwen3 filetrans dry-run 使用 file_url 单数字段", async () => { + test("speech recognize qwen3 filetrans dry-run 使用 file_url 与 language", async () => { const body = await runRecognizeDryRun([ "--model", "qwen3-asr-flash-filetrans", "--url", "https://dashscope.oss-cn-beijing.aliyuncs.com/samples/audio/paraformer/hello_world_female2.wav", + "--language", + "zh", ]); expect(body.mode).toBe("async"); expect(body.path).toBe("/api/v1/services/audio/asr/transcription"); expect(body.request?.input?.file_url?.startsWith("https://")).toBe(true); expect(body.request?.input?.file_urls).toBeUndefined(); + expect(body.request?.parameters?.language).toBe("zh"); + expect(body.request?.parameters?.language_hints).toBeUndefined(); }); test("speech recognize realtime 模型报用法错误", async () => { diff --git a/packages/core/src/client/asr-routes.ts b/packages/core/src/client/asr-routes.ts index 3297afe..4abd741 100644 --- a/packages/core/src/client/asr-routes.ts +++ b/packages/core/src/client/asr-routes.ts @@ -31,6 +31,12 @@ export interface AsrApiRoute { * - `file_url`: qwen3-asr-flash-filetrans family */ asyncInputStyle?: "file_urls" | "file_url"; + /** + * Async transcription language field style. + * - `language_hints`: fun-asr / paraformer / qwen-audio filetrans... + * - `language`: qwen3-asr-flash-filetrans* + */ + asyncLanguageStyle?: "language_hints" | "language"; flashFamily?: AsrFlashFamily; /** Human-readable reason when kind is unsupported. */ unsupportedReason?: string; @@ -90,11 +96,13 @@ export function resolveAsrApi(model: string): AsrApiRoute { } if (isFiletransModel(model)) { + const isQwen3Filetrans = isQwen3FiletransModel(model); return { kind: "async-filetrans", path: speechRecognizePath(), useSync: false, - asyncInputStyle: isQwen3FiletransModel(model) ? "file_url" : "file_urls", + asyncInputStyle: isQwen3Filetrans ? "file_url" : "file_urls", + asyncLanguageStyle: isQwen3Filetrans ? "language" : "language_hints", }; } @@ -122,6 +130,7 @@ export function resolveAsrApi(model: string): AsrApiRoute { path: speechRecognizePath(), useSync: false, asyncInputStyle: "file_urls", + asyncLanguageStyle: "language_hints", }; } @@ -139,15 +148,32 @@ export interface BuildAsrFlashRequestOpts { model: string; audioUrl: string; language?: string; + /** 预编译热词 ID;仅 input-audio Flash(fun-asr-flash* / qwen-audio-*-asr-flash)官方支持 */ + vocabularyId?: string; flashFamily: AsrFlashFamily; } +/** + * 按异步路由的 language 字段风格构造语种参数。 + * qwen3-asr-flash-filetrans* → `language`;其余异步模型 → `language_hints`。 + */ +export function buildAsyncAsrLanguageFields( + languageStyle: "language_hints" | "language", + language?: string, +): { language_hints?: string[]; language?: string } { + if (!language) return {}; + if (languageStyle === "language") { + return { language }; + } + return { language_hints: [language] }; +} + /** Build a sync multimodal ASR request body for Flash models. */ export function buildAsrFlashRequest(opts: BuildAsrFlashRequestOpts): Record { - const { model, audioUrl, language, flashFamily } = opts; + const { model, audioUrl, language, vocabularyId, flashFamily } = opts; if (flashFamily === "input-audio") { - // 与官方 Qwen-Audio / Fun-ASR-Flash 文档一致:语种走 language_hints + // 与官方 Qwen-Audio / Fun-ASR-Flash 文档一致:语种走 language_hints,热词走 vocabulary_id const parameters: Record = { format: inferAudioFormatHint(audioUrl), sample_rate: "16000", @@ -155,6 +181,9 @@ export function buildAsrFlashRequest(opts: BuildAsrFlashRequestOpts): Record { useSync: false, path: "/api/v1/services/audio/asr/transcription", asyncInputStyle: "file_url", + asyncLanguageStyle: "language", }, }, { @@ -32,6 +34,7 @@ test("resolveAsrApi routes model families correctly", () => { kind: "async-filetrans", useSync: false, asyncInputStyle: "file_urls", + asyncLanguageStyle: "language_hints", }, }, { @@ -112,6 +115,7 @@ test("buildAsrFlashRequest shapes qwen3 and input-audio bodies", () => { model: "qwen-audio-3.0-asr-flash", audioUrl: "https://example.com/a.wav", language: "en", + vocabularyId: "vocab-abc", flashFamily: "input-audio", }), ).toEqual({ @@ -124,10 +128,23 @@ test("buildAsrFlashRequest shapes qwen3 and input-audio bodies", () => { }, ], }, - parameters: { format: "wav", sample_rate: "16000", language_hints: ["en"] }, + parameters: { + format: "wav", + sample_rate: "16000", + language_hints: ["en"], + vocabulary_id: "vocab-abc", + }, }); }); +test("buildAsyncAsrLanguageFields maps language by async style", () => { + expect(buildAsyncAsrLanguageFields("language_hints", "zh")).toEqual({ + language_hints: ["zh"], + }); + expect(buildAsyncAsrLanguageFields("language", "zh")).toEqual({ language: "zh" }); + expect(buildAsyncAsrLanguageFields("language", undefined)).toEqual({}); +}); + test("extractAsrFlashText reads qwen3 choices and input-audio text fields", () => { expect( extractAsrFlashText( diff --git a/packages/runtime/src/pipeline/steps/bl-api.ts b/packages/runtime/src/pipeline/steps/bl-api.ts index b6def2d..a3ea8ca 100644 --- a/packages/runtime/src/pipeline/steps/bl-api.ts +++ b/packages/runtime/src/pipeline/steps/bl-api.ts @@ -12,6 +12,7 @@ import { speechRecognizePath, resolveAsrApi, buildAsrFlashRequest, + buildAsyncAsrLanguageFields, extractAsrFlashText, stripUndefined, resolveBooleanFlag, @@ -596,15 +597,18 @@ export async function speechRecognize( { step: "speech/recognize" }, ); } - if ( - input.diarization || - input["speaker-count"] !== undefined || - input["vocabulary-id"] !== undefined || - input["channel-id"] !== undefined - ) { + const unsupportedFlags: string[] = []; + if (input.diarization) unsupportedFlags.push("diarization"); + if (input["speaker-count"] !== undefined) unsupportedFlags.push("speaker-count"); + // input-audio Flash 官方支持 vocabulary_id;qwen3 sync Flash 不支持 + if (route.flashFamily === "qwen3" && input["vocabulary-id"] !== undefined) { + unsupportedFlags.push("vocabulary-id"); + } + if (input["channel-id"] !== undefined) unsupportedFlags.push("channel-id"); + if (unsupportedFlags.length > 0) { throw new PipelineError( "invalid_input", - `Model "${model}" uses sync Flash ASR and does not support diarization / speaker-count / vocabulary-id / channel-id`, + `Model "${model}" uses sync Flash ASR and does not support: ${unsupportedFlags.join(", ")}`, { step: "speech/recognize" }, ); } @@ -641,6 +645,7 @@ export async function speechRecognize( model, audioUrl: fileUrls[0]!, language: input.language, + vocabularyId: input["vocabulary-id"], flashFamily, }); const response = await env.client.requestJson>({ @@ -657,13 +662,17 @@ export async function speechRecognize( }; } + const languageFields = buildAsyncAsrLanguageFields( + route.asyncLanguageStyle ?? "language_hints", + input.language, + ); const body: DashScopeASRRequest = { model, input: route.asyncInputStyle === "file_url" ? { file_url: fileUrls[0]! } : { file_urls: fileUrls }, parameters: { channel_id: input["channel-id"] !== undefined ? [input["channel-id"]] : undefined, - language_hints: input.language ? [input.language] : undefined, + ...languageFields, diarization_enabled: input.diarization, speaker_count: input["speaker-count"], vocabulary_id: input["vocabulary-id"], diff --git a/skills/bailian-gen/reference/speech.md b/skills/bailian-gen/reference/speech.md index 8da88fd..5a65f35 100644 --- a/skills/bailian-gen/reference/speech.md +++ b/skills/bailian-gen/reference/speech.md @@ -28,7 +28,7 @@ Index: [index.md](index.md) | --------------------------- | ------ | -------- | ------------------------------------------------------------------------------------------------------------------------------------------- | | `--url ` | array | yes | Audio file URL or local file path (repeatable, max 100) | | `--model ` | string | no | Model ID (default: fun-asr). Async: fun-asr / _-filetrans / paraformer-_; sync: qwen3-asr-flash* / fun-asr-flash* / qwen-audio-\*-asr-flash | -| `--language ` | string | no | Language hint (e.g. zh, en, ja). Async & input-audio sync: language_hints; qwen3 sync: asr_options.language | +| `--language ` | string | no | Language hint (e.g. zh, en, ja). Classic async/input-audio: language_hints; qwen3-filetrans: language; qwen3 sync: asr_options.language | | `--diarization` | switch | no | Enable automatic speaker diarization | | `--speaker-count ` | number | no | Expected number of speakers (requires --diarization) | | `--vocabulary-id ` | string | no | Hot-word vocabulary ID for improved accuracy |