fix: support sync-flash and qwen3-filetrans ASR models in speech recognize

- Add asr-routes.ts with resolveAsrApi() to route models to the correct
  DashScope endpoint instead of always hitting asr/transcription
- Async filetrans: fun-asr / paraformer / *-filetrans → file_urls (plural)
- Async filetrans (qwen3): qwen3-asr-flash-filetrans* → file_url (singular)
- Sync flash (input-audio): fun-asr-flash* / qwen-audio-*-asr-flash → multimodal-generation
- Sync flash (qwen3): qwen3-asr-flash* → multimodal-generation + asr_options
- Realtime/streaming models now give a clear USAGE error instead of a
  confusing server-side "url error"
- Propagate same routing logic to pipeline speechRecognize step
- Add table-driven unit tests and dry-run e2e assertions
Fixes #146
This commit is contained in:
clh02467605
2026-08-12 17:12:23 +08:00
parent 2389681ad6
commit 241de61866
9 changed files with 740 additions and 50 deletions
@@ -12,6 +12,11 @@ import {
stripUndefined,
taskPath,
speechRecognizePath,
resolveAsrApi,
buildAsrFlashRequest,
extractAsrFlashText,
type AsrApiRoute,
type AsrFlashFamily,
type OutputFormat,
type FlagsDef,
type ParsedFlags,
@@ -27,8 +32,18 @@ const RECOGNIZE_FLAGS = {
description: "Audio file URL or local file path (repeatable, max 100)",
required: true,
},
model: { type: "string", valueHint: "<model>", description: "Model ID (default: fun-asr)" },
language: { type: "string", valueHint: "<lang>", description: "Language hint (e.g. zh, en, ja)" },
model: {
type: "string",
valueHint: "<model>",
description:
"Model ID (default: fun-asr). Async: fun-asr / *-filetrans / paraformer-*; sync: qwen3-asr-flash* / fun-asr-flash* / qwen-audio-*-asr-flash",
},
language: {
type: "string",
valueHint: "<lang>",
description:
"Language hint (e.g. zh, en, ja). Async & input-audio sync: language_hints; qwen3 sync: asr_options.language",
},
diarization: { type: "switch", description: "Enable automatic speaker diarization" },
speakerCount: {
type: "number",
@@ -55,8 +70,26 @@ const RECOGNIZE_FLAGS = {
} satisfies FlagsDef;
type RecognizeFlags = ParsedFlags<typeof RECOGNIZE_FLAGS>;
function assertSyncFlashFlagsAllowed(flags: RecognizeFlags, model: string): void {
const unsupported: string[] = [];
if (flags.diarization === true) unsupported.push("--diarization");
if (flags.speakerCount !== undefined) unsupported.push("--speaker-count");
if (flags.vocabularyId !== undefined) unsupported.push("--vocabulary-id");
if (flags.channelId !== undefined) unsupported.push("--channel-id");
if (flags.async === true) unsupported.push("--async");
if (flags.pollInterval !== undefined) unsupported.push("--poll-interval");
if (unsupported.length > 0) {
throw new BailianError(
`Model "${model}" uses sync Flash ASR and does not support: ${unsupported.join(", ")}.\n` +
`Hint: Use an async filetrans model (e.g. fun-asr, qwen3-asr-flash-filetrans) for those flags.`,
ExitCode.USAGE,
);
}
}
export default defineCommand({
description: "Recognize speech from audio files (FunAudio-ASR)",
description: "Recognize speech from audio files (FunAudio-ASR / Qwen-ASR Flash)",
auth: "apiKey",
usageArgs: "--url <audio-url> [flags]",
flags: RECOGNIZE_FLAGS,
@@ -68,6 +101,7 @@ export default defineCommand({
"--url https://example.com/audio.mp3 --vocabulary-id vocab-abc123",
"--url https://example.com/audio.mp3 --out result.json",
"--url https://example.com/audio.mp3 --async --quiet",
"--url https://example.com/audio.mp3 --model qwen-audio-3.0-asr-flash --language en",
],
async run(ctx) {
const { settings, flags } = ctx;
@@ -90,19 +124,64 @@ export default defineCommand({
}
const model = flags.model || "fun-asr";
const route = resolveAsrApi(model);
if (route.kind === "unsupported") {
throw new BailianError(
route.unsupportedReason ?? `Unsupported ASR model: ${model}`,
ExitCode.USAGE,
);
}
if (route.kind === "sync-flash") {
assertSyncFlashFlagsAllowed(flags, model);
if (rawUrls.length !== 1) {
throw new BailianError(
`Model "${model}" is a sync Flash ASR model and accepts exactly one --url (got ${rawUrls.length}).\n` +
`Hint: Pass a single audio URL, or use an async filetrans model for batch files.`,
ExitCode.USAGE,
);
}
}
if (
route.kind === "async-filetrans" &&
route.asyncInputStyle === "file_url" &&
rawUrls.length !== 1
) {
throw new BailianError(
`Model "${model}" accepts exactly one --url (got ${rawUrls.length}).\n` +
"Hint: qwen3-asr-flash-filetrans* requires a single file_url.",
ExitCode.USAGE,
);
}
const format = detectOutputFormat(settings.output);
// Auto-upload local files in parallel
const resolvedUrls = await Promise.all(rawUrls.map((u) => ctx.client.uploadFile(u, model)));
const resolvedUrls = await Promise.all(rawUrls.map((url) => ctx.client.uploadFile(url, model)));
if (route.kind === "sync-flash") {
await handleSyncFlashMode(
ctx.client,
settings,
flags,
format,
model,
route,
resolvedUrls[0]!,
);
return;
}
const channelId = flags.channelId;
const language = flags.language;
const vocabularyId = flags.vocabularyId;
const body: DashScopeASRRequest = {
model,
input: {
file_urls: resolvedUrls,
},
input:
route.asyncInputStyle === "file_url"
? { file_url: resolvedUrls[0]! }
: { file_urls: resolvedUrls },
parameters: {
channel_id: channelId !== undefined ? [channelId] : [0],
language_hints: language ? [language] : undefined,
@@ -116,7 +195,7 @@ export default defineCommand({
stripUndefined(body.parameters as Record<string, unknown>);
if (settings.dryRun) {
emitResult({ request: body, mode: "async" }, format);
emitResult({ request: body, mode: "async", path: speechRecognizePath() }, format);
return;
}
@@ -128,6 +207,53 @@ export default defineCommand({
},
});
async function handleSyncFlashMode(
client: Client,
settings: Settings,
flags: RecognizeFlags,
format: OutputFormat,
model: string,
route: AsrApiRoute,
audioUrl: string,
): Promise<void> {
const flashFamily = route.flashFamily as AsrFlashFamily;
const body = buildAsrFlashRequest({
model,
audioUrl,
language: flags.language,
flashFamily,
});
if (settings.dryRun) {
emitResult({ request: body, mode: "sync", path: route.path }, format);
return;
}
if (!settings.quiet) {
process.stderr.write(`[Model: ${model}] [Mode: sync] [Files: 1]\n`);
}
const response = await client.requestJson<Record<string, unknown>>({
path: route.path,
method: "POST",
body,
});
const text = extractAsrFlashText(response, flashFamily);
if (text) {
process.stdout.write(text.endsWith("\n") ? text : `${text}\n`);
} else {
emitBare(JSON.stringify(response));
}
if (flags.out) {
writeFileSync(flags.out, JSON.stringify(response, null, 2) + "\n");
if (!settings.quiet) {
process.stderr.write(`Full result saved to: ${flags.out}\n`);
}
}
}
async function handleAsyncMode(
client: Client,
settings: Settings,
@@ -160,12 +286,12 @@ async function handleAsyncMode(
url: pollUrl,
intervalSec: pollInterval,
timeoutSec: settings.timeout,
isComplete: (d) => (d as DashScopeASRTaskResult).output.task_status === "SUCCEEDED",
isFailed: (d) => (d as DashScopeASRTaskResult).output.task_status === "FAILED",
getStatus: (d) => (d as DashScopeASRTaskResult).output.task_status,
getErrorMessage: (d) => {
const o = (d as DashScopeASRTaskResult).output;
return (o as unknown as Record<string, unknown>).message as string | undefined;
isComplete: (data) => (data as DashScopeASRTaskResult).output.task_status === "SUCCEEDED",
isFailed: (data) => (data as DashScopeASRTaskResult).output.task_status === "FAILED",
getStatus: (data) => (data as DashScopeASRTaskResult).output.task_status,
getErrorMessage: (data) => {
const output = (data as DashScopeASRTaskResult).output;
return (output as unknown as Record<string, unknown>).message as string | undefined;
},
});
@@ -179,12 +305,14 @@ async function handleAsyncMode(
// Collect all transcription data for --out
const allTransData: Record<string, unknown>[] = [];
for (let i = 0; i < results.length; i++) {
const subResult = results[i]!;
for (let index = 0; index < results.length; index++) {
const subResult = results[index]!;
const isMulti = fileCount > 1;
if (isMulti) {
process.stdout.write(`=== [${i + 1}/${results.length}] ${subResult.file_url ?? ""} ===\n`);
process.stdout.write(
`=== [${index + 1}/${results.length}] ${subResult.file_url ?? ""} ===\n`,
);
}
if (subResult.subtask_status === "FAILED") {
@@ -16,6 +16,32 @@ import { SPEECH_ROUTES } from "./topic-routes.ts";
*/
describe("e2e: speech recognize", () => {
async function runRecognizeDryRun(args: string[]) {
const { stdout, stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [
"speech",
"recognize",
...args,
"--dry-run",
"--output",
"json",
"--quiet",
]);
expect(exitCode, stderr).toBe(0);
return parseStdoutJson<{
mode?: string;
path?: string;
request?: {
model?: string;
parameters?: { format?: string; language_hints?: string[] };
input?: {
file_url?: string;
file_urls?: string[];
messages?: Array<{ content?: Array<{ type?: string }> }>;
};
};
}>(stdout);
}
test("speech recognize --help 正常退出", async () => {
const { stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [
"speech",
@@ -25,6 +51,50 @@ describe("e2e: speech recognize", () => {
expect(exitCode, stderr).toBe(0);
expect(stderr).toMatch(/recognize|--url|model|audio/i);
});
test("speech recognize sync-flash dry-run 走 multimodal-generation", async () => {
const body = await runRecognizeDryRun([
"--model",
"qwen-audio-3.0-asr-flash",
"--url",
"https://dashscope.oss-cn-beijing.aliyuncs.com/samples/audio/paraformer/hello_world_female2.wav",
"--language",
"en",
]);
expect(body.mode).toBe("sync");
expect(body.path).toBe("/api/v1/services/aigc/multimodal-generation/generation");
expect(body.request?.model).toBe("qwen-audio-3.0-asr-flash");
expect(body.request?.parameters?.format).toBe("wav");
expect(body.request?.parameters?.language_hints).toEqual(["en"]);
expect(body.request?.input?.messages?.[0]?.content?.[0]?.type).toBe("input_audio");
});
test("speech recognize qwen3 filetrans dry-run 使用 file_url 单数字段", async () => {
const body = await runRecognizeDryRun([
"--model",
"qwen3-asr-flash-filetrans",
"--url",
"https://dashscope.oss-cn-beijing.aliyuncs.com/samples/audio/paraformer/hello_world_female2.wav",
]);
expect(body.mode).toBe("async");
expect(body.path).toBe("/api/v1/services/audio/asr/transcription");
expect(body.request?.input?.file_url?.startsWith("https://")).toBe(true);
expect(body.request?.input?.file_urls).toBeUndefined();
});
test("speech recognize realtime 模型报用法错误", async () => {
const { stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [
"speech",
"recognize",
"--model",
"qwen3-asr-flash-realtime",
"--url",
"https://example.com/a.wav",
"--quiet",
]);
expect(exitCode).toBe(2);
expect(stderr).toMatch(/realtime|WebSocket|unsupported/i);
});
});
describe.skipIf(!isBailianE2EMediaEnabled() || !isDashScopeE2EReady())(
+254
View File
@@ -0,0 +1,254 @@
import { imageSyncPath, speechRecognizePath } from "./endpoints.ts";
/**
* DashScope ASR APIs differ by model family:
*
* - async file transcription (`.../audio/asr/transcription`):
* fun-asr*, paraformer* (non-realtime), *-filetrans, sensevoice*
* language via `parameters.language_hints`
* - sync multimodal (`.../aigc/multimodal-generation/generation`):
* - qwen3: `{ content: [{ audio }] }` + optional `asr_options.language`
* (qwen3-asr-flash*)
* - input-audio: `{ type: input_audio, input_audio.data }` +
* `format`/`sample_rate` + optional `language_hints`
* (fun-asr-flash*, qwen-audio-*-asr-flash*)
* - realtime / streaming: WebSocket — not supported by `speech recognize`
*/
export type AsrApiKind = "async-filetrans" | "sync-flash" | "unsupported";
/** Sync-flash request body shape differs by Flash protocol family. */
export type AsrFlashFamily = "qwen3" | "input-audio";
export interface AsrApiRoute {
kind: AsrApiKind;
path: string;
/** True when the call is synchronous (no X-DashScope-Async / task poll). */
useSync: boolean;
/**
* Async transcription request input style.
* - `file_urls`: classic async models (fun-asr / paraformer / qwen-audio filetrans...)
* - `file_url`: qwen3-asr-flash-filetrans family
*/
asyncInputStyle?: "file_urls" | "file_url";
flashFamily?: AsrFlashFamily;
/** Human-readable reason when kind is unsupported. */
unsupportedReason?: string;
}
function isRealtimeOrStreaming(model: string): boolean {
return /realtime|streaming/i.test(model);
}
function isFiletransModel(model: string): boolean {
return /filetrans/i.test(model);
}
function isQwen3FiletransModel(model: string): boolean {
return /^qwen3-asr-flash-filetrans(?:-|$)/i.test(model);
}
const INPUT_AUDIO_FLASH_PREFIXES = ["fun-asr-flash", "qwen-audio"] as const;
/**
* Fun-ASR-Flash / Qwen-Audio-*-ASR-Flash share the input_audio + format protocol.
* Examples: fun-asr-flash-2026-06-15, qwen-audio-3.0-asr-flash
*/
function isInputAudioFlashModel(model: string): boolean {
if (isRealtimeOrStreaming(model) || isFiletransModel(model)) return false;
if (model.startsWith(INPUT_AUDIO_FLASH_PREFIXES[0])) return true;
if (model.startsWith(INPUT_AUDIO_FLASH_PREFIXES[1]) && /asr-flash/i.test(model)) return true;
return false;
}
/**
* Qwen3-ASR-Flash sync models use content.audio + asr_options.
* Examples: qwen3-asr-flash, qwen3-asr-flash-2025-09-08, qwen3-asr-flash-us
*/
function isQwen3AsrFlashModel(model: string): boolean {
if (!/^qwen3-asr-flash(?:-|$)/i.test(model)) return false;
if (isFiletransModel(model) || isRealtimeOrStreaming(model)) return false;
if (isInputAudioFlashModel(model)) return false;
return true;
}
/**
* Resolve which DashScope ASR API a model should use for file recognition.
* Unknown models default to async-filetrans (preserves existing CLI behavior).
*/
export function resolveAsrApi(model: string): AsrApiRoute {
if (isRealtimeOrStreaming(model)) {
return {
kind: "unsupported",
path: "",
useSync: false,
unsupportedReason:
`Model "${model}" is a realtime/streaming ASR model and requires a WebSocket API. ` +
`Use an async filetrans model (e.g. fun-asr, qwen3-asr-flash-filetrans) or a sync flash model ` +
`(e.g. qwen3-asr-flash, qwen-audio-3.0-asr-flash) with this command.`,
};
}
if (isFiletransModel(model)) {
return {
kind: "async-filetrans",
path: speechRecognizePath(),
useSync: false,
asyncInputStyle: isQwen3FiletransModel(model) ? "file_url" : "file_urls",
};
}
if (isInputAudioFlashModel(model)) {
return {
kind: "sync-flash",
path: imageSyncPath(),
useSync: true,
flashFamily: "input-audio",
};
}
if (isQwen3AsrFlashModel(model)) {
return {
kind: "sync-flash",
path: imageSyncPath(),
useSync: true,
flashFamily: "qwen3",
};
}
// fun-asr / paraformer / sensevoice / unknown → keep legacy async path
return {
kind: "async-filetrans",
path: speechRecognizePath(),
useSync: false,
asyncInputStyle: "file_urls",
};
}
/** Infer audio container hint for input-audio Flash `parameters.format`. */
export function inferAudioFormatHint(audioUrl: string): string {
const pathPart = audioUrl.split("?")[0] ?? audioUrl;
const match = pathPart.match(/\.([a-zA-Z0-9]+)$/);
const extension = match?.[1]?.toLowerCase();
if (!extension) return "wav";
if (extension === "mpeg") return "mp3";
return extension;
}
export interface BuildAsrFlashRequestOpts {
model: string;
audioUrl: string;
language?: string;
flashFamily: AsrFlashFamily;
}
/** Build a sync multimodal ASR request body for Flash models. */
export function buildAsrFlashRequest(opts: BuildAsrFlashRequestOpts): Record<string, unknown> {
const { model, audioUrl, language, flashFamily } = opts;
if (flashFamily === "input-audio") {
// 与官方 Qwen-Audio / Fun-ASR-Flash 文档一致:语种走 language_hints
const parameters: Record<string, unknown> = {
format: inferAudioFormatHint(audioUrl),
sample_rate: "16000",
};
if (language) {
parameters.language_hints = [language];
}
return {
model,
input: {
messages: [
{
role: "user",
content: [
{
type: "input_audio",
input_audio: { data: audioUrl },
},
],
},
],
},
parameters,
};
}
const asrOptions: Record<string, unknown> = {};
if (language) {
asrOptions.language = language;
}
const parameters: Record<string, unknown> = {};
if (Object.keys(asrOptions).length > 0) {
parameters.asr_options = asrOptions;
}
const body: Record<string, unknown> = {
model,
input: {
messages: [
{
role: "user",
content: [{ audio: audioUrl }],
},
],
},
};
if (Object.keys(parameters).length > 0) {
body.parameters = parameters;
}
return body;
}
/**
* Extract recognition text from a sync Flash ASR response.
* Qwen3 uses choices[].message.content; input-audio Flash uses output.text.
*/
export function extractAsrFlashText(
response: Record<string, unknown>,
flashFamily: AsrFlashFamily,
): string {
const output = response.output as Record<string, unknown> | undefined;
if (!output) return "";
if (flashFamily === "input-audio") {
if (typeof output.text === "string" && output.text.length > 0) {
return output.text;
}
const nested = output.output as Record<string, unknown> | undefined;
const sentence = nested?.sentence as Record<string, unknown> | undefined;
if (typeof sentence?.text === "string") {
return sentence.text;
}
return "";
}
const choices = output.choices as Array<Record<string, unknown>> | undefined;
if (!choices?.length) return "";
const texts: string[] = [];
for (const choice of choices) {
const message = choice.message as Record<string, unknown> | undefined;
if (!message) continue;
const content = message.content;
if (typeof content === "string") {
texts.push(content);
continue;
}
if (!Array.isArray(content)) continue;
for (const item of content) {
if (typeof item === "string") {
texts.push(item);
continue;
}
if (item && typeof item === "object") {
const record = item as Record<string, unknown>;
if (typeof record.text === "string") {
texts.push(record.text);
}
}
}
}
return texts.join("");
}
+10
View File
@@ -34,6 +34,16 @@ export {
type ImageInputStyle,
type ImageSizeProfile,
} from "./image-routes.ts";
export {
buildAsrFlashRequest,
extractAsrFlashText,
inferAudioFormatHint,
resolveAsrApi,
type AsrApiKind,
type AsrApiRoute,
type AsrFlashFamily,
type BuildAsrFlashRequestOpts,
} from "./asr-routes.ts";
export { CHANNEL, sourceConfig, trackingHeaders, type TrackingIdentity } from "./headers.ts";
export type { HttpDeps, RequestOpts } from "./http.ts";
export { request, requestJson } from "./http.ts";
+2 -1
View File
@@ -533,7 +533,8 @@ export interface DashScopeTTSStreamChunk {
export interface DashScopeASRRequest {
model: string;
input: {
file_urls: string[];
file_urls?: string[];
file_url?: string;
};
parameters?: {
channel_id?: number[];
+154
View File
@@ -0,0 +1,154 @@
import { expect, test } from "vite-plus/test";
import {
buildAsrFlashRequest,
extractAsrFlashText,
inferAudioFormatHint,
resolveAsrApi,
} from "../src/client/asr-routes.ts";
test("resolveAsrApi routes model families correctly", () => {
const cases = [
{
model: "fun-asr",
expected: {
kind: "async-filetrans",
useSync: false,
path: "/api/v1/services/audio/asr/transcription",
asyncInputStyle: "file_urls",
},
},
{
model: "qwen3-asr-flash-filetrans-2025-11-17",
expected: {
kind: "async-filetrans",
useSync: false,
path: "/api/v1/services/audio/asr/transcription",
asyncInputStyle: "file_url",
},
},
{
model: "qwen-audio-3.0-asr-flash-filetrans",
expected: {
kind: "async-filetrans",
useSync: false,
asyncInputStyle: "file_urls",
},
},
{
model: "qwen3-asr-flash-us",
expected: {
kind: "sync-flash",
useSync: true,
flashFamily: "qwen3",
path: "/api/v1/services/aigc/multimodal-generation/generation",
},
},
{
model: "qwen-audio-3.0-asr-flash",
expected: {
kind: "sync-flash",
useSync: true,
flashFamily: "input-audio",
},
},
{
model: "qwen3-asr-flash-realtime",
expected: {
kind: "unsupported",
},
},
{
model: "foo-asr-flash",
expected: {
kind: "async-filetrans",
useSync: false,
path: "/api/v1/services/audio/asr/transcription",
asyncInputStyle: "file_urls",
},
},
] as const;
for (const { model, expected } of cases) {
const route = resolveAsrApi(model);
expect(route, model).toMatchObject(expected);
if (expected.kind === "unsupported") {
expect(route.unsupportedReason, model).toMatch(/realtime|streaming|WebSocket/i);
}
}
});
test("unknown models default to async-filetrans for backward compatibility", () => {
expect(resolveAsrApi("custom-asr-model")).toMatchObject({
kind: "async-filetrans",
useSync: false,
});
});
test("inferAudioFormatHint reads extension from url", () => {
expect(inferAudioFormatHint("https://example.com/a.mp3")).toBe("mp3");
expect(inferAudioFormatHint("oss://bucket/path/file.WAV")).toBe("wav");
expect(inferAudioFormatHint("https://example.com/a.mpeg?x=1")).toBe("mp3");
expect(inferAudioFormatHint("https://example.com/noext")).toBe("wav");
});
test("buildAsrFlashRequest shapes qwen3 and input-audio bodies", () => {
expect(
buildAsrFlashRequest({
model: "qwen3-asr-flash",
audioUrl: "https://example.com/a.mp3",
language: "en",
flashFamily: "qwen3",
}),
).toEqual({
model: "qwen3-asr-flash",
input: {
messages: [{ role: "user", content: [{ audio: "https://example.com/a.mp3" }] }],
},
parameters: { asr_options: { language: "en" } },
});
expect(
buildAsrFlashRequest({
model: "qwen-audio-3.0-asr-flash",
audioUrl: "https://example.com/a.wav",
language: "en",
flashFamily: "input-audio",
}),
).toEqual({
model: "qwen-audio-3.0-asr-flash",
input: {
messages: [
{
role: "user",
content: [{ type: "input_audio", input_audio: { data: "https://example.com/a.wav" } }],
},
],
},
parameters: { format: "wav", sample_rate: "16000", language_hints: ["en"] },
});
});
test("extractAsrFlashText reads qwen3 choices and input-audio text fields", () => {
expect(
extractAsrFlashText(
{
output: {
choices: [{ message: { content: [{ text: "你好" }] } }],
},
},
"qwen3",
),
).toBe("你好");
expect(
extractAsrFlashText(
{
output: {
text: "Hello World",
output: { sentence: { text: "ignored when text present" } },
},
},
"input-audio",
),
).toBe("Hello World");
});
+77 -8
View File
@@ -10,6 +10,9 @@ import {
taskPath,
speechSynthesizePath,
speechRecognizePath,
resolveAsrApi,
buildAsrFlashRequest,
extractAsrFlashText,
stripUndefined,
resolveBooleanFlag,
resolveWatermark,
@@ -573,24 +576,91 @@ export async function speechRecognize(
});
}
const model = input.model || "fun-asr";
const route = resolveAsrApi(model);
if (route.kind === "unsupported") {
throw new PipelineError(
"invalid_input",
route.unsupportedReason ?? `Unsupported ASR model: ${model}`,
{
step: "speech/recognize",
},
);
}
if (route.kind === "sync-flash") {
if (rawUrls.length !== 1) {
throw new PipelineError(
"invalid_input",
`Model "${model}" is a sync Flash ASR model and accepts exactly one url (got ${rawUrls.length})`,
{ step: "speech/recognize" },
);
}
if (
input.diarization ||
input["speaker-count"] !== undefined ||
input["vocabulary-id"] !== undefined ||
input["channel-id"] !== undefined
) {
throw new PipelineError(
"invalid_input",
`Model "${model}" uses sync Flash ASR and does not support diarization / speaker-count / vocabulary-id / channel-id`,
{ step: "speech/recognize" },
);
}
}
if (
route.kind === "async-filetrans" &&
route.asyncInputStyle === "file_url" &&
rawUrls.length !== 1
) {
throw new PipelineError(
"invalid_input",
`Model "${model}" accepts exactly one url (got ${rawUrls.length})`,
{ step: "speech/recognize" },
);
}
// Resolve local files to upload URLs
const fileUrls: string[] = [];
for (const u of rawUrls) {
if (isLocalFile(u)) {
for (const audioUrl of rawUrls) {
if (isLocalFile(audioUrl)) {
fileUrls.push(
await env.client.uploadFile(u, input.model || "fun-asr", {
await env.client.uploadFile(audioUrl, model, {
signal: ctx.signal,
}),
);
} else {
fileUrls.push(u);
fileUrls.push(audioUrl);
}
}
const model = input.model || "fun-asr";
if (route.kind === "sync-flash") {
const flashFamily = route.flashFamily!;
const body = buildAsrFlashRequest({
model,
audioUrl: fileUrls[0]!,
language: input.language,
flashFamily,
});
const response = await env.client.requestJson<Record<string, unknown>>({
path: route.path,
method: "POST",
body,
signal: ctx.signal,
});
return {
text: extractAsrFlashText(response, flashFamily),
model,
mode: "sync",
raw: response,
};
}
const body: DashScopeASRRequest = {
model,
input: { file_urls: fileUrls },
input:
route.asyncInputStyle === "file_url" ? { file_url: fileUrls[0]! } : { file_urls: fileUrls },
parameters: {
channel_id: input["channel-id"] !== undefined ? [input["channel-id"]] : undefined,
language_hints: input.language ? [input.language] : undefined,
@@ -601,9 +671,8 @@ export async function speechRecognize(
};
stripUndefined(body.parameters as Record<string, unknown>);
const url = speechRecognizePath();
const asyncResp = await env.client.requestJson<DashScopeAsyncResponse>({
path: url,
path: speechRecognizePath(),
method: "POST",
body,
async: true,
+1 -1
View File
@@ -14,7 +14,7 @@ Use this index for the skill-scoped quick index and global flags.
| `bl image edit` | Edit an existing image with text instructions (Qwen-Image / Wan 2.7) | [image.md](image.md) |
| `bl image generate` | Generate images (Qwen-Image / wan2.x) | [image.md](image.md) |
| `bl omni` | Multimodal chat with text + audio output (Qwen-Omni) | [omni.md](omni.md) |
| `bl speech recognize` | Recognize speech from audio files (FunAudio-ASR) | [speech.md](speech.md) |
| `bl speech recognize` | Recognize speech from audio files (FunAudio-ASR / Qwen-ASR Flash) | [speech.md](speech.md) |
| `bl speech synthesize` | Synthesize speech from text (CosyVoice TTS) | [speech.md](speech.md) |
| `bl video download` | Download a completed video by task ID | [video.md](video.md) |
| `bl video edit` | Edit a video with happyhorse-1.0-video-edit (style transfer, object replacement, etc.) | [video.md](video.md) |
+27 -23
View File
@@ -7,37 +7,37 @@ Index: [index.md](index.md)
## Commands in this group
| Command | Description |
| ---------------------- | ------------------------------------------------ |
| `bl speech recognize` | Recognize speech from audio files (FunAudio-ASR) |
| `bl speech synthesize` | Synthesize speech from text (CosyVoice TTS) |
| Command | Description |
| ---------------------- | ----------------------------------------------------------------- |
| `bl speech recognize` | Recognize speech from audio files (FunAudio-ASR / Qwen-ASR Flash) |
| `bl speech synthesize` | Synthesize speech from text (CosyVoice TTS) |
## Command details
### `bl speech recognize`
| Field | Value |
| --------------- | ------------------------------------------------ |
| **Name** | `speech recognize` |
| **Description** | Recognize speech from audio files (FunAudio-ASR) |
| **Usage** | `bl speech recognize --url <audio-url> [flags]` |
| Field | Value |
| --------------- | ----------------------------------------------------------------- |
| **Name** | `speech recognize` |
| **Description** | Recognize speech from audio files (FunAudio-ASR / Qwen-ASR Flash) |
| **Usage** | `bl speech recognize --url <audio-url> [flags]` |
#### Flags
| Flag | Type | Required | Description |
| --------------------------- | ------ | -------- | ------------------------------------------------------- |
| `--url <url>` | array | yes | Audio file URL or local file path (repeatable, max 100) |
| `--model <model>` | string | no | Model ID (default: fun-asr) |
| `--language <lang>` | string | no | Language hint (e.g. zh, en, ja) |
| `--diarization` | switch | no | Enable automatic speaker diarization |
| `--speaker-count <n>` | number | no | Expected number of speakers (requires --diarization) |
| `--vocabulary-id <id>` | string | no | Hot-word vocabulary ID for improved accuracy |
| `--channel-id <n>` | number | no | Audio channel ID (default: 0) |
| `--out <path>` | string | no | Save full transcription result to JSON file |
| `--async` | switch | no | Return async task id without waiting |
| `--poll-interval <seconds>` | number | no | Polling interval in seconds (default: 2) |
| `--api-key <key>` | string | no | API key |
| `--base-url <url>` | string | no | API base URL |
| Flag | Type | Required | Description |
| --------------------------- | ------ | -------- | ------------------------------------------------------------------------------------------------------------------------------------------- |
| `--url <url>` | array | yes | Audio file URL or local file path (repeatable, max 100) |
| `--model <model>` | string | no | Model ID (default: fun-asr). Async: fun-asr / _-filetrans / paraformer-_; sync: qwen3-asr-flash* / fun-asr-flash* / qwen-audio-\*-asr-flash |
| `--language <lang>` | string | no | Language hint (e.g. zh, en, ja). Async & input-audio sync: language_hints; qwen3 sync: asr_options.language |
| `--diarization` | switch | no | Enable automatic speaker diarization |
| `--speaker-count <n>` | number | no | Expected number of speakers (requires --diarization) |
| `--vocabulary-id <id>` | string | no | Hot-word vocabulary ID for improved accuracy |
| `--channel-id <n>` | number | no | Audio channel ID (default: 0) |
| `--out <path>` | string | no | Save full transcription result to JSON file |
| `--async` | switch | no | Return async task id without waiting |
| `--poll-interval <seconds>` | number | no | Polling interval in seconds (default: 2) |
| `--api-key <key>` | string | no | API key |
| `--base-url <url>` | string | no | API base URL |
#### Examples
@@ -69,6 +69,10 @@ bl speech recognize --url https://example.com/audio.mp3 --out result.json
bl speech recognize --url https://example.com/audio.mp3 --async --quiet
```
```bash
bl speech recognize --url https://example.com/audio.mp3 --model qwen-audio-3.0-asr-flash --language en
```
### `bl speech synthesize`
| Field | Value |