mirror of
https://github.com/modelstudioai/cli.git
synced 2026-09-14 19:49:23 +08:00
Merge pull request #149 from modelstudioai/fix/fixed_issue_146
fix: support sync-flash and qwen3-filetrans ASR models in speech recognize
This commit is contained in:
@@ -12,6 +12,13 @@ import {
|
||||
stripUndefined,
|
||||
taskPath,
|
||||
speechRecognizePath,
|
||||
resolveAsrApi,
|
||||
buildAsrFlashRequest,
|
||||
buildAsyncAsrLanguageFields,
|
||||
collectAsrTranscriptionItems,
|
||||
extractAsrFlashText,
|
||||
type AsrApiRoute,
|
||||
type AsrFlashFamily,
|
||||
type OutputFormat,
|
||||
type FlagsDef,
|
||||
type ParsedFlags,
|
||||
@@ -27,8 +34,18 @@ const RECOGNIZE_FLAGS = {
|
||||
description: "Audio file URL or local file path (repeatable, max 100)",
|
||||
required: true,
|
||||
},
|
||||
model: { type: "string", valueHint: "<model>", description: "Model ID (default: fun-asr)" },
|
||||
language: { type: "string", valueHint: "<lang>", description: "Language hint (e.g. zh, en, ja)" },
|
||||
model: {
|
||||
type: "string",
|
||||
valueHint: "<model>",
|
||||
description:
|
||||
"Model ID (default: fun-asr). Async: fun-asr / *-filetrans / paraformer-*; sync: qwen3-asr-flash* / fun-asr-flash* / qwen-audio-*-asr-flash",
|
||||
},
|
||||
language: {
|
||||
type: "string",
|
||||
valueHint: "<lang>",
|
||||
description:
|
||||
"Language hint (e.g. zh, en, ja). Classic async/input-audio: language_hints; qwen3-filetrans: language; qwen3 sync: asr_options.language",
|
||||
},
|
||||
diarization: { type: "switch", description: "Enable automatic speaker diarization" },
|
||||
speakerCount: {
|
||||
type: "number",
|
||||
@@ -55,8 +72,33 @@ const RECOGNIZE_FLAGS = {
|
||||
} satisfies FlagsDef;
|
||||
type RecognizeFlags = ParsedFlags<typeof RECOGNIZE_FLAGS>;
|
||||
|
||||
function assertSyncFlashFlagsAllowed(
|
||||
flags: RecognizeFlags,
|
||||
model: string,
|
||||
flashFamily: AsrFlashFamily,
|
||||
): void {
|
||||
const unsupported: string[] = [];
|
||||
if (flags.diarization === true) unsupported.push("--diarization");
|
||||
if (flags.speakerCount !== undefined) unsupported.push("--speaker-count");
|
||||
// qwen3 sync Flash does not use vocabulary_id; input-audio Flash (fun-asr-flash* / qwen-audio-*-asr-flash) does
|
||||
if (flashFamily === "qwen3" && flags.vocabularyId !== undefined) {
|
||||
unsupported.push("--vocabulary-id");
|
||||
}
|
||||
if (flags.channelId !== undefined) unsupported.push("--channel-id");
|
||||
if (flags.async === true) unsupported.push("--async");
|
||||
if (flags.pollInterval !== undefined) unsupported.push("--poll-interval");
|
||||
|
||||
if (unsupported.length > 0) {
|
||||
throw new BailianError(
|
||||
`Model "${model}" uses sync Flash ASR and does not support: ${unsupported.join(", ")}.\n` +
|
||||
`Hint: Use an async filetrans model (e.g. fun-asr, qwen3-asr-flash-filetrans) for those flags.`,
|
||||
ExitCode.USAGE,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
export default defineCommand({
|
||||
description: "Recognize speech from audio files (FunAudio-ASR)",
|
||||
description: "Recognize speech from audio files (FunAudio-ASR / Qwen-ASR Flash)",
|
||||
auth: "apiKey",
|
||||
usageArgs: "--url <audio-url> [flags]",
|
||||
flags: RECOGNIZE_FLAGS,
|
||||
@@ -68,6 +110,7 @@ export default defineCommand({
|
||||
"--url https://example.com/audio.mp3 --vocabulary-id vocab-abc123",
|
||||
"--url https://example.com/audio.mp3 --out result.json",
|
||||
"--url https://example.com/audio.mp3 --async --quiet",
|
||||
"--url https://example.com/audio.mp3 --model qwen-audio-3.0-asr-flash --language en",
|
||||
],
|
||||
async run(ctx) {
|
||||
const { settings, flags } = ctx;
|
||||
@@ -90,22 +133,70 @@ export default defineCommand({
|
||||
}
|
||||
|
||||
const model = flags.model || "fun-asr";
|
||||
const route = resolveAsrApi(model);
|
||||
if (route.kind === "unsupported") {
|
||||
throw new BailianError(
|
||||
route.unsupportedReason ?? `Unsupported ASR model: ${model}`,
|
||||
ExitCode.USAGE,
|
||||
);
|
||||
}
|
||||
|
||||
if (route.kind === "sync-flash") {
|
||||
assertSyncFlashFlagsAllowed(flags, model, route.flashFamily!);
|
||||
if (rawUrls.length !== 1) {
|
||||
throw new BailianError(
|
||||
`Model "${model}" is a sync Flash ASR model and accepts exactly one --url (got ${rawUrls.length}).\n` +
|
||||
`Hint: Pass a single audio URL, or use an async filetrans model for batch files.`,
|
||||
ExitCode.USAGE,
|
||||
);
|
||||
}
|
||||
}
|
||||
if (
|
||||
route.kind === "async-filetrans" &&
|
||||
route.asyncInputStyle === "file_url" &&
|
||||
rawUrls.length !== 1
|
||||
) {
|
||||
throw new BailianError(
|
||||
`Model "${model}" accepts exactly one --url (got ${rawUrls.length}).\n` +
|
||||
"Hint: qwen3-asr-flash-filetrans* requires a single file_url.",
|
||||
ExitCode.USAGE,
|
||||
);
|
||||
}
|
||||
|
||||
const format = detectOutputFormat(settings.output);
|
||||
|
||||
// Auto-upload local files in parallel
|
||||
const resolvedUrls = await Promise.all(rawUrls.map((u) => ctx.client.uploadFile(u, model)));
|
||||
const resolvedUrls = await Promise.all(rawUrls.map((url) => ctx.client.uploadFile(url, model)));
|
||||
|
||||
if (route.kind === "sync-flash") {
|
||||
await handleSyncFlashMode(
|
||||
ctx.client,
|
||||
settings,
|
||||
flags,
|
||||
format,
|
||||
model,
|
||||
route,
|
||||
resolvedUrls[0]!,
|
||||
);
|
||||
return;
|
||||
}
|
||||
|
||||
const channelId = flags.channelId;
|
||||
const language = flags.language;
|
||||
const vocabularyId = flags.vocabularyId;
|
||||
const languageFields = buildAsyncAsrLanguageFields(
|
||||
route.asyncLanguageStyle ?? "language_hints",
|
||||
flags.language,
|
||||
);
|
||||
|
||||
const body: DashScopeASRRequest = {
|
||||
model,
|
||||
input: {
|
||||
file_urls: resolvedUrls,
|
||||
},
|
||||
input:
|
||||
route.asyncInputStyle === "file_url"
|
||||
? { file_url: resolvedUrls[0]! }
|
||||
: { file_urls: resolvedUrls },
|
||||
parameters: {
|
||||
channel_id: channelId !== undefined ? [channelId] : [0],
|
||||
language_hints: language ? [language] : undefined,
|
||||
...languageFields,
|
||||
diarization_enabled: diarization ? true : undefined,
|
||||
speaker_count: speakerCount,
|
||||
vocabulary_id: vocabularyId,
|
||||
@@ -116,7 +207,7 @@ export default defineCommand({
|
||||
stripUndefined(body.parameters as Record<string, unknown>);
|
||||
|
||||
if (settings.dryRun) {
|
||||
emitResult({ request: body, mode: "async" }, format);
|
||||
emitResult({ request: body, mode: "async", path: speechRecognizePath() }, format);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -128,6 +219,55 @@ export default defineCommand({
|
||||
},
|
||||
});
|
||||
|
||||
async function handleSyncFlashMode(
|
||||
client: Client,
|
||||
settings: Settings,
|
||||
flags: RecognizeFlags,
|
||||
format: OutputFormat,
|
||||
model: string,
|
||||
route: AsrApiRoute,
|
||||
audioUrl: string,
|
||||
): Promise<void> {
|
||||
const flashFamily = route.flashFamily as AsrFlashFamily;
|
||||
const body = buildAsrFlashRequest({
|
||||
model,
|
||||
audioUrl,
|
||||
language: flags.language,
|
||||
vocabularyId: flags.vocabularyId,
|
||||
flashFamily,
|
||||
});
|
||||
|
||||
if (settings.dryRun) {
|
||||
emitResult({ request: body, mode: "sync", path: route.path }, format);
|
||||
return;
|
||||
}
|
||||
|
||||
if (!settings.quiet) {
|
||||
process.stderr.write(`[Model: ${model}] [Mode: sync] [Files: 1]\n`);
|
||||
}
|
||||
|
||||
const response = await client.requestJson<Record<string, unknown>>({
|
||||
path: route.path,
|
||||
method: "POST",
|
||||
headers: { "X-DashScope-SSE": "disable" },
|
||||
body,
|
||||
});
|
||||
|
||||
const text = extractAsrFlashText(response, flashFamily);
|
||||
if (text) {
|
||||
process.stdout.write(text.endsWith("\n") ? text : `${text}\n`);
|
||||
} else {
|
||||
emitBare(JSON.stringify(response));
|
||||
}
|
||||
|
||||
if (flags.out) {
|
||||
writeFileSync(flags.out, JSON.stringify(response, null, 2) + "\n");
|
||||
if (!settings.quiet) {
|
||||
process.stderr.write(`Full result saved to: ${flags.out}\n`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async function handleAsyncMode(
|
||||
client: Client,
|
||||
settings: Settings,
|
||||
@@ -160,16 +300,16 @@ async function handleAsyncMode(
|
||||
url: pollUrl,
|
||||
intervalSec: pollInterval,
|
||||
timeoutSec: settings.timeout,
|
||||
isComplete: (d) => (d as DashScopeASRTaskResult).output.task_status === "SUCCEEDED",
|
||||
isFailed: (d) => (d as DashScopeASRTaskResult).output.task_status === "FAILED",
|
||||
getStatus: (d) => (d as DashScopeASRTaskResult).output.task_status,
|
||||
getErrorMessage: (d) => {
|
||||
const o = (d as DashScopeASRTaskResult).output;
|
||||
return (o as unknown as Record<string, unknown>).message as string | undefined;
|
||||
isComplete: (data) => (data as DashScopeASRTaskResult).output.task_status === "SUCCEEDED",
|
||||
isFailed: (data) => (data as DashScopeASRTaskResult).output.task_status === "FAILED",
|
||||
getStatus: (data) => (data as DashScopeASRTaskResult).output.task_status,
|
||||
getErrorMessage: (data) => {
|
||||
const output = (data as DashScopeASRTaskResult).output;
|
||||
return (output as unknown as Record<string, unknown>).message as string | undefined;
|
||||
},
|
||||
});
|
||||
|
||||
const results = result.output.results ?? [];
|
||||
const results = collectAsrTranscriptionItems(result.output);
|
||||
|
||||
if (results.length === 0) {
|
||||
emitResult({ task_id: taskId, status: result.output.task_status }, format);
|
||||
@@ -179,12 +319,14 @@ async function handleAsyncMode(
|
||||
// Collect all transcription data for --out
|
||||
const allTransData: Record<string, unknown>[] = [];
|
||||
|
||||
for (let i = 0; i < results.length; i++) {
|
||||
const subResult = results[i]!;
|
||||
for (let index = 0; index < results.length; index++) {
|
||||
const subResult = results[index]!;
|
||||
const isMulti = fileCount > 1;
|
||||
|
||||
if (isMulti) {
|
||||
process.stdout.write(`=== [${i + 1}/${results.length}] ${subResult.file_url ?? ""} ===\n`);
|
||||
process.stdout.write(
|
||||
`=== [${index + 1}/${results.length}] ${subResult.file_url ?? ""} ===\n`,
|
||||
);
|
||||
}
|
||||
|
||||
if (subResult.subtask_status === "FAILED") {
|
||||
|
||||
@@ -1,4 +1,6 @@
|
||||
import { readFileSync } from "node:fs";
|
||||
import http from "node:http";
|
||||
import type { AddressInfo } from "node:net";
|
||||
import { join } from "node:path";
|
||||
import { describe, expect, test } from "vite-plus/test";
|
||||
import {
|
||||
@@ -16,6 +18,37 @@ import { SPEECH_ROUTES } from "./topic-routes.ts";
|
||||
*/
|
||||
|
||||
describe("e2e: speech recognize", () => {
|
||||
async function runRecognizeDryRun(args: string[]) {
|
||||
const { stdout, stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [
|
||||
"speech",
|
||||
"recognize",
|
||||
...args,
|
||||
"--dry-run",
|
||||
"--output",
|
||||
"json",
|
||||
"--quiet",
|
||||
]);
|
||||
expect(exitCode, stderr).toBe(0);
|
||||
return parseStdoutJson<{
|
||||
mode?: string;
|
||||
path?: string;
|
||||
request?: {
|
||||
model?: string;
|
||||
parameters?: {
|
||||
format?: string;
|
||||
language_hints?: string[];
|
||||
language?: string;
|
||||
vocabulary_id?: string;
|
||||
};
|
||||
input?: {
|
||||
file_url?: string;
|
||||
file_urls?: string[];
|
||||
messages?: Array<{ content?: Array<{ type?: string }> }>;
|
||||
};
|
||||
};
|
||||
}>(stdout);
|
||||
}
|
||||
|
||||
test("speech recognize --help 正常退出", async () => {
|
||||
const { stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [
|
||||
"speech",
|
||||
@@ -25,6 +58,217 @@ describe("e2e: speech recognize", () => {
|
||||
expect(exitCode, stderr).toBe(0);
|
||||
expect(stderr).toMatch(/recognize|--url|model|audio/i);
|
||||
});
|
||||
|
||||
test("speech recognize sync-flash dry-run 走 multimodal-generation", async () => {
|
||||
const body = await runRecognizeDryRun([
|
||||
"--model",
|
||||
"qwen-audio-3.0-asr-flash",
|
||||
"--url",
|
||||
"https://dashscope.oss-cn-beijing.aliyuncs.com/samples/audio/paraformer/hello_world_female2.wav",
|
||||
"--language",
|
||||
"en",
|
||||
"--vocabulary-id",
|
||||
"vocab-e2e",
|
||||
]);
|
||||
expect(body.mode).toBe("sync");
|
||||
expect(body.path).toBe("/api/v1/services/aigc/multimodal-generation/generation");
|
||||
expect(body.request?.model).toBe("qwen-audio-3.0-asr-flash");
|
||||
expect(body.request?.parameters?.format).toBe("wav");
|
||||
expect(body.request?.parameters?.language_hints).toEqual(["en"]);
|
||||
expect(body.request?.parameters?.vocabulary_id).toBe("vocab-e2e");
|
||||
expect(body.request?.input?.messages?.[0]?.content?.[0]?.type).toBe("input_audio");
|
||||
});
|
||||
|
||||
test("speech recognize qwen3 filetrans dry-run 使用 file_url 与 language", async () => {
|
||||
const body = await runRecognizeDryRun([
|
||||
"--model",
|
||||
"qwen3-asr-flash-filetrans",
|
||||
"--url",
|
||||
"https://dashscope.oss-cn-beijing.aliyuncs.com/samples/audio/paraformer/hello_world_female2.wav",
|
||||
"--language",
|
||||
"zh",
|
||||
]);
|
||||
expect(body.mode).toBe("async");
|
||||
expect(body.path).toBe("/api/v1/services/audio/asr/transcription");
|
||||
expect(body.request?.input?.file_url?.startsWith("https://")).toBe(true);
|
||||
expect(body.request?.input?.file_urls).toBeUndefined();
|
||||
expect(body.request?.parameters?.language).toBe("zh");
|
||||
expect(body.request?.parameters?.language_hints).toBeUndefined();
|
||||
});
|
||||
|
||||
test("speech recognize realtime 模型报用法错误", async () => {
|
||||
// Use --dry-run to skip auth so CI without API keys still hits USAGE(2)
|
||||
const { stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [
|
||||
"speech",
|
||||
"recognize",
|
||||
"--model",
|
||||
"qwen3-asr-flash-realtime",
|
||||
"--url",
|
||||
"https://example.com/a.wav",
|
||||
"--dry-run",
|
||||
"--quiet",
|
||||
]);
|
||||
expect(exitCode).toBe(2);
|
||||
expect(stderr).toMatch(/realtime|WebSocket|unsupported/i);
|
||||
});
|
||||
|
||||
test("speech recognize flash 真实请求走 sync endpoint 并落盘 --out", async () => {
|
||||
let requestPath = "";
|
||||
let requestBody: Record<string, unknown> = {};
|
||||
let sseHeader: string | undefined;
|
||||
const server = http.createServer((request, response) => {
|
||||
const chunks: Buffer[] = [];
|
||||
request.on("data", (chunk: Buffer) => chunks.push(chunk));
|
||||
request.on("end", () => {
|
||||
requestPath = request.url ?? "";
|
||||
requestBody = JSON.parse(Buffer.concat(chunks).toString("utf8")) as Record<string, unknown>;
|
||||
sseHeader = request.headers["x-dashscope-sse"] as string | undefined;
|
||||
response.writeHead(200, { "Content-Type": "application/json" });
|
||||
response.end(
|
||||
JSON.stringify({
|
||||
output: { text: "flash recognition works" },
|
||||
request_id: "request-146",
|
||||
}),
|
||||
);
|
||||
});
|
||||
});
|
||||
await new Promise<void>((resolve) => server.listen(0, "127.0.0.1", resolve));
|
||||
const address = server.address() as AddressInfo;
|
||||
const outDir = makeE2eOutputDir("speech-recognize-flash-sync");
|
||||
const outPath = join(outDir, "result.json");
|
||||
|
||||
try {
|
||||
const { stdout, stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [
|
||||
"speech",
|
||||
"recognize",
|
||||
"--model",
|
||||
"fun-asr-flash-2026-06-15",
|
||||
"--url",
|
||||
"https://example.com/sample.wav",
|
||||
"--api-key",
|
||||
"sk-e2e-placeholder",
|
||||
"--base-url",
|
||||
`http://127.0.0.1:${address.port}`,
|
||||
"--out",
|
||||
outPath,
|
||||
"--quiet",
|
||||
]);
|
||||
|
||||
expect(exitCode, stderr).toBe(0);
|
||||
expect(stdout).toContain("flash recognition works");
|
||||
expect(requestPath).toBe("/api/v1/services/aigc/multimodal-generation/generation");
|
||||
expect(sseHeader).toBe("disable");
|
||||
expect(requestBody).toMatchObject({
|
||||
model: "fun-asr-flash-2026-06-15",
|
||||
parameters: { format: "wav" },
|
||||
});
|
||||
expect(JSON.parse(readFileSync(outPath, "utf8"))).toMatchObject({
|
||||
output: { text: "flash recognition works" },
|
||||
request_id: "request-146",
|
||||
});
|
||||
} finally {
|
||||
await new Promise<void>((resolve) => server.close(() => resolve()));
|
||||
}
|
||||
});
|
||||
|
||||
test("speech recognize flash 多 --url 在发请求前报用法错误", async () => {
|
||||
const { stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [
|
||||
"speech",
|
||||
"recognize",
|
||||
"--model",
|
||||
"qwen-audio-3.0-asr-flash",
|
||||
"--url",
|
||||
"https://example.com/a.wav",
|
||||
"--url",
|
||||
"https://example.com/b.wav",
|
||||
"--dry-run",
|
||||
"--quiet",
|
||||
]);
|
||||
|
||||
expect(exitCode).toBe(2);
|
||||
expect(stderr).toMatch(/exactly one --url|sync Flash/i);
|
||||
});
|
||||
|
||||
test("speech recognize qwen3-filetrans 轮询成功后下载 result.transcription_url", async () => {
|
||||
const server = http.createServer((request, response) => {
|
||||
const url = request.url ?? "";
|
||||
const chunks: Buffer[] = [];
|
||||
request.on("data", (chunk: Buffer) => chunks.push(chunk));
|
||||
request.on("end", () => {
|
||||
response.writeHead(200, { "Content-Type": "application/json" });
|
||||
if (url.startsWith("/api/v1/services/audio/asr/transcription")) {
|
||||
response.end(
|
||||
JSON.stringify({
|
||||
output: { task_id: "task-qwen3", task_status: "PENDING" },
|
||||
request_id: "req-submit",
|
||||
}),
|
||||
);
|
||||
return;
|
||||
}
|
||||
if (url.startsWith("/api/v1/tasks/")) {
|
||||
const address = server.address() as AddressInfo;
|
||||
response.end(
|
||||
JSON.stringify({
|
||||
output: {
|
||||
task_id: "task-qwen3",
|
||||
task_status: "SUCCEEDED",
|
||||
result: {
|
||||
transcription_url: `http://127.0.0.1:${address.port}/transcription.json`,
|
||||
},
|
||||
},
|
||||
request_id: "req-poll",
|
||||
}),
|
||||
);
|
||||
return;
|
||||
}
|
||||
if (url.startsWith("/transcription.json")) {
|
||||
response.end(
|
||||
JSON.stringify({
|
||||
file_url: "https://example.com/a.wav",
|
||||
transcripts: [{ text: "你好世界", sentences: [{ text: "你好世界" }] }],
|
||||
}),
|
||||
);
|
||||
return;
|
||||
}
|
||||
response.writeHead(404);
|
||||
response.end(JSON.stringify({ message: `unexpected path: ${url}` }));
|
||||
});
|
||||
});
|
||||
await new Promise<void>((resolve) => server.listen(0, "127.0.0.1", resolve));
|
||||
const address = server.address() as AddressInfo;
|
||||
const outDir = makeE2eOutputDir("speech-recognize-qwen3-filetrans");
|
||||
const outPath = join(outDir, "result.json");
|
||||
|
||||
try {
|
||||
const { stdout, stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [
|
||||
"speech",
|
||||
"recognize",
|
||||
"--model",
|
||||
"qwen3-asr-flash-filetrans",
|
||||
"--url",
|
||||
"https://example.com/a.wav",
|
||||
"--language",
|
||||
"zh",
|
||||
"--api-key",
|
||||
"sk-e2e-placeholder",
|
||||
"--base-url",
|
||||
`http://127.0.0.1:${address.port}`,
|
||||
"--poll-interval",
|
||||
"1",
|
||||
"--out",
|
||||
outPath,
|
||||
"--quiet",
|
||||
]);
|
||||
|
||||
expect(exitCode, stderr).toBe(0);
|
||||
expect(stdout).toContain("你好世界");
|
||||
expect(JSON.parse(readFileSync(outPath, "utf8"))).toMatchObject({
|
||||
transcripts: [{ text: "你好世界" }],
|
||||
});
|
||||
} finally {
|
||||
await new Promise<void>((resolve) => server.close(() => resolve()));
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe.skipIf(!isBailianE2EMediaEnabled() || !isDashScopeE2EReady())(
|
||||
|
||||
@@ -0,0 +1,327 @@
|
||||
import { imageSyncPath, speechRecognizePath } from "./endpoints.ts";
|
||||
|
||||
/**
|
||||
* DashScope ASR APIs differ by model family:
|
||||
*
|
||||
* - async file transcription (`.../audio/asr/transcription`):
|
||||
* fun-asr*, paraformer* (non-realtime), *-filetrans, sensevoice*
|
||||
* language via `parameters.language_hints`
|
||||
* - sync multimodal (`.../aigc/multimodal-generation/generation`):
|
||||
* - qwen3: `{ content: [{ audio }] }` + optional `asr_options.language`
|
||||
* (qwen3-asr-flash*)
|
||||
* - input-audio: `{ type: input_audio, input_audio.data }` +
|
||||
* `format`/`sample_rate` + optional `language_hints`
|
||||
* (fun-asr-flash*, qwen-audio-*-asr-flash*)
|
||||
* - realtime / streaming: WebSocket — not supported by `speech recognize`
|
||||
*/
|
||||
|
||||
export type AsrApiKind = "async-filetrans" | "sync-flash" | "unsupported";
|
||||
|
||||
/** Sync-flash request body shape differs by Flash protocol family. */
|
||||
export type AsrFlashFamily = "qwen3" | "input-audio";
|
||||
|
||||
export interface AsrApiRoute {
|
||||
kind: AsrApiKind;
|
||||
path: string;
|
||||
/** True when the call is synchronous (no X-DashScope-Async / task poll). */
|
||||
useSync: boolean;
|
||||
/**
|
||||
* Async transcription request input style.
|
||||
* - `file_urls`: classic async models (fun-asr / paraformer / qwen-audio filetrans...)
|
||||
* - `file_url`: qwen3-asr-flash-filetrans family
|
||||
*/
|
||||
asyncInputStyle?: "file_urls" | "file_url";
|
||||
/**
|
||||
* Async transcription language field style.
|
||||
* - `language_hints`: fun-asr / paraformer / qwen-audio filetrans...
|
||||
* - `language`: qwen3-asr-flash-filetrans*
|
||||
*/
|
||||
asyncLanguageStyle?: "language_hints" | "language";
|
||||
flashFamily?: AsrFlashFamily;
|
||||
/** Human-readable reason when kind is unsupported. */
|
||||
unsupportedReason?: string;
|
||||
}
|
||||
|
||||
function isRealtimeOrStreaming(model: string): boolean {
|
||||
return /realtime|streaming/i.test(model);
|
||||
}
|
||||
|
||||
function isFiletransModel(model: string): boolean {
|
||||
return /filetrans/i.test(model);
|
||||
}
|
||||
|
||||
function isQwen3FiletransModel(model: string): boolean {
|
||||
return /^qwen3-asr-flash-filetrans(?:-|$)/i.test(model);
|
||||
}
|
||||
|
||||
const INPUT_AUDIO_FLASH_PREFIXES = ["fun-asr-flash", "qwen-audio"] as const;
|
||||
|
||||
/**
|
||||
* Fun-ASR-Flash / Qwen-Audio-*-ASR-Flash share the input_audio + format protocol.
|
||||
* Examples: fun-asr-flash-2026-06-15, qwen-audio-3.0-asr-flash
|
||||
*/
|
||||
function isInputAudioFlashModel(model: string): boolean {
|
||||
if (isRealtimeOrStreaming(model) || isFiletransModel(model)) return false;
|
||||
if (model.startsWith(INPUT_AUDIO_FLASH_PREFIXES[0])) return true;
|
||||
if (model.startsWith(INPUT_AUDIO_FLASH_PREFIXES[1]) && /asr-flash/i.test(model)) return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Qwen3-ASR-Flash sync models use content.audio + asr_options.
|
||||
* Examples: qwen3-asr-flash, qwen3-asr-flash-2025-09-08, qwen3-asr-flash-us
|
||||
*/
|
||||
function isQwen3AsrFlashModel(model: string): boolean {
|
||||
if (!/^qwen3-asr-flash(?:-|$)/i.test(model)) return false;
|
||||
if (isFiletransModel(model) || isRealtimeOrStreaming(model)) return false;
|
||||
if (isInputAudioFlashModel(model)) return false;
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve which DashScope ASR API a model should use for file recognition.
|
||||
* Unknown models default to async-filetrans (preserves existing CLI behavior).
|
||||
*/
|
||||
export function resolveAsrApi(model: string): AsrApiRoute {
|
||||
if (isRealtimeOrStreaming(model)) {
|
||||
return {
|
||||
kind: "unsupported",
|
||||
path: "",
|
||||
useSync: false,
|
||||
unsupportedReason:
|
||||
`Model "${model}" is a realtime/streaming ASR model and requires a WebSocket API. ` +
|
||||
`Use an async filetrans model (e.g. fun-asr, qwen3-asr-flash-filetrans) or a sync flash model ` +
|
||||
`(e.g. qwen3-asr-flash, qwen-audio-3.0-asr-flash) with this command.`,
|
||||
};
|
||||
}
|
||||
|
||||
if (isFiletransModel(model)) {
|
||||
const isQwen3Filetrans = isQwen3FiletransModel(model);
|
||||
return {
|
||||
kind: "async-filetrans",
|
||||
path: speechRecognizePath(),
|
||||
useSync: false,
|
||||
asyncInputStyle: isQwen3Filetrans ? "file_url" : "file_urls",
|
||||
asyncLanguageStyle: isQwen3Filetrans ? "language" : "language_hints",
|
||||
};
|
||||
}
|
||||
|
||||
if (isInputAudioFlashModel(model)) {
|
||||
return {
|
||||
kind: "sync-flash",
|
||||
path: imageSyncPath(),
|
||||
useSync: true,
|
||||
flashFamily: "input-audio",
|
||||
};
|
||||
}
|
||||
|
||||
if (isQwen3AsrFlashModel(model)) {
|
||||
return {
|
||||
kind: "sync-flash",
|
||||
path: imageSyncPath(),
|
||||
useSync: true,
|
||||
flashFamily: "qwen3",
|
||||
};
|
||||
}
|
||||
|
||||
// fun-asr / paraformer / sensevoice / unknown → keep legacy async path
|
||||
return {
|
||||
kind: "async-filetrans",
|
||||
path: speechRecognizePath(),
|
||||
useSync: false,
|
||||
asyncInputStyle: "file_urls",
|
||||
asyncLanguageStyle: "language_hints",
|
||||
};
|
||||
}
|
||||
|
||||
/** Infer audio container hint for input-audio Flash `parameters.format`. */
|
||||
export function inferAudioFormatHint(audioUrl: string): string {
|
||||
// data URI: data:audio/mpeg;base64,... → mp3; data:audio/x-wav;... → wav
|
||||
const dataType = /^data:audio\/([^;,]+)/i.exec(audioUrl)?.[1]?.toLowerCase();
|
||||
if (dataType) {
|
||||
if (dataType === "mpeg") return "mp3";
|
||||
if (dataType === "x-wav" || dataType === "wave") return "wav";
|
||||
return dataType;
|
||||
}
|
||||
|
||||
const pathPart = audioUrl.split(/[?#]/, 1)[0] ?? audioUrl;
|
||||
const match = pathPart.match(/\.([a-zA-Z0-9]+)$/);
|
||||
const extension = match?.[1]?.toLowerCase();
|
||||
if (!extension) return "wav";
|
||||
if (extension === "mpeg") return "mp3";
|
||||
return extension;
|
||||
}
|
||||
|
||||
export interface BuildAsrFlashRequestOpts {
|
||||
model: string;
|
||||
audioUrl: string;
|
||||
language?: string;
|
||||
/** Precompiled hotword vocabulary ID; supported for input-audio Flash (fun-asr-flash* / qwen-audio-*-asr-flash). */
|
||||
vocabularyId?: string;
|
||||
flashFamily: AsrFlashFamily;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build language fields for async ASR routes.
|
||||
* qwen3-asr-flash-filetrans* → `language`; other async models → `language_hints`.
|
||||
*/
|
||||
export function buildAsyncAsrLanguageFields(
|
||||
languageStyle: "language_hints" | "language",
|
||||
language?: string,
|
||||
): { language_hints?: string[]; language?: string } {
|
||||
if (!language) return {};
|
||||
if (languageStyle === "language") {
|
||||
return { language };
|
||||
}
|
||||
return { language_hints: [language] };
|
||||
}
|
||||
|
||||
/** Build a sync multimodal ASR request body for Flash models. */
|
||||
export function buildAsrFlashRequest(opts: BuildAsrFlashRequestOpts): Record<string, unknown> {
|
||||
const { model, audioUrl, language, vocabularyId, flashFamily } = opts;
|
||||
|
||||
if (flashFamily === "input-audio") {
|
||||
// Match official Qwen-Audio / Fun-ASR-Flash docs: language_hints + vocabulary_id
|
||||
const parameters: Record<string, unknown> = {
|
||||
format: inferAudioFormatHint(audioUrl),
|
||||
sample_rate: "16000",
|
||||
};
|
||||
if (language) {
|
||||
parameters.language_hints = [language];
|
||||
}
|
||||
if (vocabularyId) {
|
||||
parameters.vocabulary_id = vocabularyId;
|
||||
}
|
||||
return {
|
||||
model,
|
||||
input: {
|
||||
messages: [
|
||||
{
|
||||
role: "user",
|
||||
content: [
|
||||
{
|
||||
type: "input_audio",
|
||||
input_audio: { data: audioUrl },
|
||||
},
|
||||
],
|
||||
},
|
||||
],
|
||||
},
|
||||
parameters,
|
||||
};
|
||||
}
|
||||
|
||||
const asrOptions: Record<string, unknown> = {};
|
||||
if (language) {
|
||||
asrOptions.language = language;
|
||||
}
|
||||
|
||||
const parameters: Record<string, unknown> = {};
|
||||
if (Object.keys(asrOptions).length > 0) {
|
||||
parameters.asr_options = asrOptions;
|
||||
}
|
||||
|
||||
const body: Record<string, unknown> = {
|
||||
model,
|
||||
input: {
|
||||
messages: [
|
||||
{
|
||||
role: "user",
|
||||
content: [{ audio: audioUrl }],
|
||||
},
|
||||
],
|
||||
},
|
||||
};
|
||||
if (Object.keys(parameters).length > 0) {
|
||||
body.parameters = parameters;
|
||||
}
|
||||
return body;
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract recognition text from a sync Flash ASR response.
|
||||
* Qwen3 uses choices[].message.content; input-audio Flash uses output.text /
|
||||
* output.sentence.text / output.output.sentence.text.
|
||||
*/
|
||||
export function extractAsrFlashText(
|
||||
response: Record<string, unknown>,
|
||||
flashFamily: AsrFlashFamily,
|
||||
): string {
|
||||
const output = response.output as Record<string, unknown> | undefined;
|
||||
if (!output) return "";
|
||||
|
||||
if (flashFamily === "input-audio") {
|
||||
if (typeof output.text === "string" && output.text.length > 0) {
|
||||
return output.text;
|
||||
}
|
||||
const topSentence = output.sentence as Record<string, unknown> | undefined;
|
||||
if (typeof topSentence?.text === "string" && topSentence.text.length > 0) {
|
||||
return topSentence.text;
|
||||
}
|
||||
const nested = output.output as Record<string, unknown> | undefined;
|
||||
const nestedSentence = nested?.sentence as Record<string, unknown> | undefined;
|
||||
if (typeof nestedSentence?.text === "string") {
|
||||
return nestedSentence.text;
|
||||
}
|
||||
return "";
|
||||
}
|
||||
|
||||
const choices = output.choices as Array<Record<string, unknown>> | undefined;
|
||||
if (!choices?.length) return "";
|
||||
|
||||
const texts: string[] = [];
|
||||
for (const choice of choices) {
|
||||
const message = choice.message as Record<string, unknown> | undefined;
|
||||
if (!message) continue;
|
||||
const content = message.content;
|
||||
if (typeof content === "string") {
|
||||
texts.push(content);
|
||||
continue;
|
||||
}
|
||||
if (!Array.isArray(content)) continue;
|
||||
for (const item of content) {
|
||||
if (typeof item === "string") {
|
||||
texts.push(item);
|
||||
continue;
|
||||
}
|
||||
if (item && typeof item === "object") {
|
||||
const record = item as Record<string, unknown>;
|
||||
if (typeof record.text === "string") {
|
||||
texts.push(record.text);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return texts.join("");
|
||||
}
|
||||
|
||||
/**
|
||||
* Normalize async ASR task transcription items:
|
||||
* - classic models: `output.results[]`
|
||||
* - qwen3-asr-flash-filetrans*: `output.result.transcription_url`
|
||||
*/
|
||||
export function collectAsrTranscriptionItems(output: {
|
||||
results?: Array<{
|
||||
file_url?: string;
|
||||
transcription_url?: string;
|
||||
subtask_status?: string;
|
||||
code?: string;
|
||||
message?: string;
|
||||
}>;
|
||||
result?: { transcription_url?: string };
|
||||
}): Array<{
|
||||
file_url?: string;
|
||||
transcription_url?: string;
|
||||
subtask_status?: string;
|
||||
code?: string;
|
||||
message?: string;
|
||||
}> {
|
||||
if (output.results && output.results.length > 0) {
|
||||
return output.results;
|
||||
}
|
||||
const transcriptionUrl = output.result?.transcription_url;
|
||||
if (typeof transcriptionUrl === "string" && transcriptionUrl.length > 0) {
|
||||
return [{ transcription_url: transcriptionUrl, subtask_status: "SUCCEEDED" }];
|
||||
}
|
||||
return [];
|
||||
}
|
||||
@@ -34,6 +34,18 @@ export {
|
||||
type ImageInputStyle,
|
||||
type ImageSizeProfile,
|
||||
} from "./image-routes.ts";
|
||||
export {
|
||||
buildAsrFlashRequest,
|
||||
buildAsyncAsrLanguageFields,
|
||||
collectAsrTranscriptionItems,
|
||||
extractAsrFlashText,
|
||||
inferAudioFormatHint,
|
||||
resolveAsrApi,
|
||||
type AsrApiKind,
|
||||
type AsrApiRoute,
|
||||
type AsrFlashFamily,
|
||||
type BuildAsrFlashRequestOpts,
|
||||
} from "./asr-routes.ts";
|
||||
export { CHANNEL, sourceConfig, trackingHeaders, type TrackingIdentity } from "./headers.ts";
|
||||
export type { HttpDeps, RequestOpts } from "./http.ts";
|
||||
export { request, requestJson } from "./http.ts";
|
||||
|
||||
@@ -533,33 +533,46 @@ export interface DashScopeTTSStreamChunk {
|
||||
export interface DashScopeASRRequest {
|
||||
model: string;
|
||||
input: {
|
||||
file_urls: string[];
|
||||
file_urls?: string[];
|
||||
file_url?: string;
|
||||
};
|
||||
parameters?: {
|
||||
channel_id?: number[];
|
||||
/** Classic async models (fun-asr / paraformer / qwen-audio filetrans, etc.) */
|
||||
language_hints?: string[];
|
||||
/** qwen3-asr-flash-filetrans* uses singular `language` */
|
||||
language?: string;
|
||||
diarization_enabled?: boolean;
|
||||
speaker_count?: number;
|
||||
vocabulary_id?: string;
|
||||
};
|
||||
}
|
||||
|
||||
export interface DashScopeASRTranscriptionItem {
|
||||
file_url?: string;
|
||||
transcription_url?: string;
|
||||
subtask_status?: string;
|
||||
code?: string;
|
||||
message?: string;
|
||||
}
|
||||
|
||||
export interface DashScopeASRTaskResult {
|
||||
output: {
|
||||
task_id: string;
|
||||
task_status: "PENDING" | "RUNNING" | "SUCCEEDED" | "FAILED" | "UNKNOWN";
|
||||
results?: Array<{
|
||||
file_url?: string;
|
||||
/** Multi-file async results (fun-asr / paraformer / qwen-audio filetrans, etc.) */
|
||||
results?: DashScopeASRTranscriptionItem[];
|
||||
/** Singular result returned by qwen3-asr-flash-filetrans* on success */
|
||||
result?: {
|
||||
transcription_url?: string;
|
||||
subtask_status?: string;
|
||||
code?: string;
|
||||
message?: string;
|
||||
}>;
|
||||
};
|
||||
task_metrics?: {
|
||||
TOTAL: number;
|
||||
SUCCEEDED: number;
|
||||
FAILED: number;
|
||||
};
|
||||
code?: string;
|
||||
message?: string;
|
||||
};
|
||||
usage?: Record<string, unknown>;
|
||||
request_id: string;
|
||||
|
||||
@@ -48,6 +48,7 @@ export type {
|
||||
ChatTool,
|
||||
DashScopeASRRequest,
|
||||
DashScopeASRTaskResult,
|
||||
DashScopeASRTranscriptionItem,
|
||||
DashScopeAsyncResponse,
|
||||
DashScopeImageRequest,
|
||||
DashScopeImageSyncResponse,
|
||||
|
||||
@@ -0,0 +1,213 @@
|
||||
import { expect, test } from "vite-plus/test";
|
||||
import {
|
||||
buildAsrFlashRequest,
|
||||
buildAsyncAsrLanguageFields,
|
||||
collectAsrTranscriptionItems,
|
||||
extractAsrFlashText,
|
||||
inferAudioFormatHint,
|
||||
resolveAsrApi,
|
||||
} from "../src/client/asr-routes.ts";
|
||||
|
||||
test("resolveAsrApi routes model families correctly", () => {
|
||||
const cases = [
|
||||
{
|
||||
model: "fun-asr",
|
||||
expected: {
|
||||
kind: "async-filetrans",
|
||||
useSync: false,
|
||||
path: "/api/v1/services/audio/asr/transcription",
|
||||
asyncInputStyle: "file_urls",
|
||||
},
|
||||
},
|
||||
{
|
||||
model: "qwen3-asr-flash-filetrans-2025-11-17",
|
||||
expected: {
|
||||
kind: "async-filetrans",
|
||||
useSync: false,
|
||||
path: "/api/v1/services/audio/asr/transcription",
|
||||
asyncInputStyle: "file_url",
|
||||
asyncLanguageStyle: "language",
|
||||
},
|
||||
},
|
||||
{
|
||||
model: "qwen-audio-3.0-asr-flash-filetrans",
|
||||
expected: {
|
||||
kind: "async-filetrans",
|
||||
useSync: false,
|
||||
asyncInputStyle: "file_urls",
|
||||
asyncLanguageStyle: "language_hints",
|
||||
},
|
||||
},
|
||||
{
|
||||
model: "qwen3-asr-flash-us",
|
||||
expected: {
|
||||
kind: "sync-flash",
|
||||
useSync: true,
|
||||
flashFamily: "qwen3",
|
||||
path: "/api/v1/services/aigc/multimodal-generation/generation",
|
||||
},
|
||||
},
|
||||
{
|
||||
model: "qwen-audio-3.0-asr-flash",
|
||||
expected: {
|
||||
kind: "sync-flash",
|
||||
useSync: true,
|
||||
flashFamily: "input-audio",
|
||||
},
|
||||
},
|
||||
{
|
||||
model: "qwen3-asr-flash-realtime",
|
||||
expected: {
|
||||
kind: "unsupported",
|
||||
},
|
||||
},
|
||||
{
|
||||
model: "foo-asr-flash",
|
||||
expected: {
|
||||
kind: "async-filetrans",
|
||||
useSync: false,
|
||||
path: "/api/v1/services/audio/asr/transcription",
|
||||
asyncInputStyle: "file_urls",
|
||||
},
|
||||
},
|
||||
] as const;
|
||||
|
||||
for (const { model, expected } of cases) {
|
||||
const route = resolveAsrApi(model);
|
||||
expect(route, model).toMatchObject(expected);
|
||||
if (expected.kind === "unsupported") {
|
||||
expect(route.unsupportedReason, model).toMatch(/realtime|streaming|WebSocket/i);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
test("unknown models default to async-filetrans for backward compatibility", () => {
|
||||
expect(resolveAsrApi("custom-asr-model")).toMatchObject({
|
||||
kind: "async-filetrans",
|
||||
useSync: false,
|
||||
});
|
||||
});
|
||||
|
||||
test("inferAudioFormatHint reads extension from url", () => {
|
||||
expect(inferAudioFormatHint("https://example.com/a.mp3")).toBe("mp3");
|
||||
expect(inferAudioFormatHint("oss://bucket/path/file.WAV")).toBe("wav");
|
||||
expect(inferAudioFormatHint("https://example.com/a.mpeg?x=1")).toBe("mp3");
|
||||
expect(inferAudioFormatHint("https://example.com/noext")).toBe("wav");
|
||||
expect(inferAudioFormatHint("data:audio/mpeg;base64,AAA")).toBe("mp3");
|
||||
expect(inferAudioFormatHint("data:audio/x-wav;base64,AAA")).toBe("wav");
|
||||
expect(inferAudioFormatHint("data:audio/ogg;codecs=opus;base64,AAA")).toBe("ogg");
|
||||
});
|
||||
|
||||
test("buildAsrFlashRequest shapes qwen3 and input-audio bodies", () => {
|
||||
expect(
|
||||
buildAsrFlashRequest({
|
||||
model: "qwen3-asr-flash",
|
||||
audioUrl: "https://example.com/a.mp3",
|
||||
language: "en",
|
||||
flashFamily: "qwen3",
|
||||
}),
|
||||
).toEqual({
|
||||
model: "qwen3-asr-flash",
|
||||
input: {
|
||||
messages: [{ role: "user", content: [{ audio: "https://example.com/a.mp3" }] }],
|
||||
},
|
||||
parameters: { asr_options: { language: "en" } },
|
||||
});
|
||||
|
||||
expect(
|
||||
buildAsrFlashRequest({
|
||||
model: "qwen-audio-3.0-asr-flash",
|
||||
audioUrl: "https://example.com/a.wav",
|
||||
language: "en",
|
||||
vocabularyId: "vocab-abc",
|
||||
flashFamily: "input-audio",
|
||||
}),
|
||||
).toEqual({
|
||||
model: "qwen-audio-3.0-asr-flash",
|
||||
input: {
|
||||
messages: [
|
||||
{
|
||||
role: "user",
|
||||
content: [{ type: "input_audio", input_audio: { data: "https://example.com/a.wav" } }],
|
||||
},
|
||||
],
|
||||
},
|
||||
parameters: {
|
||||
format: "wav",
|
||||
sample_rate: "16000",
|
||||
language_hints: ["en"],
|
||||
vocabulary_id: "vocab-abc",
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
test("buildAsyncAsrLanguageFields maps language by async style", () => {
|
||||
expect(buildAsyncAsrLanguageFields("language_hints", "zh")).toEqual({
|
||||
language_hints: ["zh"],
|
||||
});
|
||||
expect(buildAsyncAsrLanguageFields("language", "zh")).toEqual({ language: "zh" });
|
||||
expect(buildAsyncAsrLanguageFields("language", undefined)).toEqual({});
|
||||
});
|
||||
|
||||
test("extractAsrFlashText reads qwen3 choices and input-audio text fields", () => {
|
||||
expect(
|
||||
extractAsrFlashText(
|
||||
{
|
||||
output: {
|
||||
choices: [{ message: { content: [{ text: "你好" }] } }],
|
||||
},
|
||||
},
|
||||
"qwen3",
|
||||
),
|
||||
).toBe("你好");
|
||||
|
||||
expect(
|
||||
extractAsrFlashText(
|
||||
{
|
||||
output: {
|
||||
text: "Hello World",
|
||||
output: { sentence: { text: "ignored when text present" } },
|
||||
},
|
||||
},
|
||||
"input-audio",
|
||||
),
|
||||
).toBe("Hello World");
|
||||
|
||||
expect(
|
||||
extractAsrFlashText(
|
||||
{
|
||||
output: {
|
||||
sentence: { text: "top-level sentence" },
|
||||
},
|
||||
},
|
||||
"input-audio",
|
||||
),
|
||||
).toBe("top-level sentence");
|
||||
|
||||
expect(
|
||||
extractAsrFlashText(
|
||||
{
|
||||
output: {
|
||||
output: { sentence: { text: "nested sentence" } },
|
||||
},
|
||||
},
|
||||
"input-audio",
|
||||
),
|
||||
).toBe("nested sentence");
|
||||
});
|
||||
|
||||
test("collectAsrTranscriptionItems prefers results[] then singular result", () => {
|
||||
expect(
|
||||
collectAsrTranscriptionItems({
|
||||
results: [{ transcription_url: "https://example.com/a.json", file_url: "https://a.wav" }],
|
||||
}),
|
||||
).toEqual([{ transcription_url: "https://example.com/a.json", file_url: "https://a.wav" }]);
|
||||
|
||||
expect(
|
||||
collectAsrTranscriptionItems({
|
||||
result: { transcription_url: "https://example.com/qwen3.json" },
|
||||
}),
|
||||
).toEqual([{ transcription_url: "https://example.com/qwen3.json", subtask_status: "SUCCEEDED" }]);
|
||||
|
||||
expect(collectAsrTranscriptionItems({})).toEqual([]);
|
||||
});
|
||||
@@ -10,6 +10,11 @@ import {
|
||||
taskPath,
|
||||
speechSynthesizePath,
|
||||
speechRecognizePath,
|
||||
resolveAsrApi,
|
||||
buildAsrFlashRequest,
|
||||
buildAsyncAsrLanguageFields,
|
||||
collectAsrTranscriptionItems,
|
||||
extractAsrFlashText,
|
||||
stripUndefined,
|
||||
resolveBooleanFlag,
|
||||
resolveWatermark,
|
||||
@@ -23,6 +28,7 @@ import {
|
||||
type DashScopeTTSRequest,
|
||||
type DashScopeTTSResponse,
|
||||
type DashScopeASRRequest,
|
||||
type DashScopeASRTaskResult,
|
||||
type ChatMessageContent,
|
||||
isLocalFile,
|
||||
} from "bailian-cli-core";
|
||||
@@ -573,27 +579,103 @@ export async function speechRecognize(
|
||||
});
|
||||
}
|
||||
|
||||
const model = input.model || "fun-asr";
|
||||
const route = resolveAsrApi(model);
|
||||
if (route.kind === "unsupported") {
|
||||
throw new PipelineError(
|
||||
"invalid_input",
|
||||
route.unsupportedReason ?? `Unsupported ASR model: ${model}`,
|
||||
{
|
||||
step: "speech/recognize",
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
if (route.kind === "sync-flash") {
|
||||
if (rawUrls.length !== 1) {
|
||||
throw new PipelineError(
|
||||
"invalid_input",
|
||||
`Model "${model}" is a sync Flash ASR model and accepts exactly one url (got ${rawUrls.length})`,
|
||||
{ step: "speech/recognize" },
|
||||
);
|
||||
}
|
||||
const unsupportedFlags: string[] = [];
|
||||
if (input.diarization) unsupportedFlags.push("diarization");
|
||||
if (input["speaker-count"] !== undefined) unsupportedFlags.push("speaker-count");
|
||||
// input-audio Flash supports vocabulary_id; qwen3 sync Flash does not
|
||||
if (route.flashFamily === "qwen3" && input["vocabulary-id"] !== undefined) {
|
||||
unsupportedFlags.push("vocabulary-id");
|
||||
}
|
||||
if (input["channel-id"] !== undefined) unsupportedFlags.push("channel-id");
|
||||
if (unsupportedFlags.length > 0) {
|
||||
throw new PipelineError(
|
||||
"invalid_input",
|
||||
`Model "${model}" uses sync Flash ASR and does not support: ${unsupportedFlags.join(", ")}`,
|
||||
{ step: "speech/recognize" },
|
||||
);
|
||||
}
|
||||
}
|
||||
if (
|
||||
route.kind === "async-filetrans" &&
|
||||
route.asyncInputStyle === "file_url" &&
|
||||
rawUrls.length !== 1
|
||||
) {
|
||||
throw new PipelineError(
|
||||
"invalid_input",
|
||||
`Model "${model}" accepts exactly one url (got ${rawUrls.length})`,
|
||||
{ step: "speech/recognize" },
|
||||
);
|
||||
}
|
||||
|
||||
// Resolve local files to upload URLs
|
||||
const fileUrls: string[] = [];
|
||||
for (const u of rawUrls) {
|
||||
if (isLocalFile(u)) {
|
||||
for (const audioUrl of rawUrls) {
|
||||
if (isLocalFile(audioUrl)) {
|
||||
fileUrls.push(
|
||||
await env.client.uploadFile(u, input.model || "fun-asr", {
|
||||
await env.client.uploadFile(audioUrl, model, {
|
||||
signal: ctx.signal,
|
||||
}),
|
||||
);
|
||||
} else {
|
||||
fileUrls.push(u);
|
||||
fileUrls.push(audioUrl);
|
||||
}
|
||||
}
|
||||
|
||||
const model = input.model || "fun-asr";
|
||||
if (route.kind === "sync-flash") {
|
||||
const flashFamily = route.flashFamily!;
|
||||
const body = buildAsrFlashRequest({
|
||||
model,
|
||||
audioUrl: fileUrls[0]!,
|
||||
language: input.language,
|
||||
vocabularyId: input["vocabulary-id"],
|
||||
flashFamily,
|
||||
});
|
||||
const response = await env.client.requestJson<Record<string, unknown>>({
|
||||
path: route.path,
|
||||
method: "POST",
|
||||
headers: { "X-DashScope-SSE": "disable" },
|
||||
body,
|
||||
signal: ctx.signal,
|
||||
});
|
||||
return {
|
||||
text: extractAsrFlashText(response, flashFamily),
|
||||
model,
|
||||
mode: "sync",
|
||||
raw: response,
|
||||
};
|
||||
}
|
||||
|
||||
const languageFields = buildAsyncAsrLanguageFields(
|
||||
route.asyncLanguageStyle ?? "language_hints",
|
||||
input.language,
|
||||
);
|
||||
const body: DashScopeASRRequest = {
|
||||
model,
|
||||
input: { file_urls: fileUrls },
|
||||
input:
|
||||
route.asyncInputStyle === "file_url" ? { file_url: fileUrls[0]! } : { file_urls: fileUrls },
|
||||
parameters: {
|
||||
channel_id: input["channel-id"] !== undefined ? [input["channel-id"]] : undefined,
|
||||
language_hints: input.language ? [input.language] : undefined,
|
||||
...languageFields,
|
||||
diarization_enabled: input.diarization,
|
||||
speaker_count: input["speaker-count"],
|
||||
vocabulary_id: input["vocabulary-id"],
|
||||
@@ -601,9 +683,8 @@ export async function speechRecognize(
|
||||
};
|
||||
stripUndefined(body.parameters as Record<string, unknown>);
|
||||
|
||||
const url = speechRecognizePath();
|
||||
const asyncResp = await env.client.requestJson<DashScopeAsyncResponse>({
|
||||
path: url,
|
||||
path: speechRecognizePath(),
|
||||
method: "POST",
|
||||
body,
|
||||
async: true,
|
||||
@@ -614,7 +695,65 @@ export async function speechRecognize(
|
||||
const pollIntervalMs = (input["poll-interval"] ?? 2) * 1000;
|
||||
const timeoutMs = (ctx.timeoutSeconds ?? 300) * 1000;
|
||||
|
||||
return await pollTaskWithOptions(env, taskId, pollIntervalMs, timeoutMs, ctx);
|
||||
// ASR polling reads original output, avoids generic flatten (avoids transcription_url polluting media urls)
|
||||
const asrTask = await pollAsrTaskWithOptions(env, taskId, pollIntervalMs, timeoutMs, ctx);
|
||||
const transcriptionItems = collectAsrTranscriptionItems(asrTask.output);
|
||||
|
||||
const base: Record<string, unknown> = {
|
||||
task_id: asrTask.output.task_id,
|
||||
task_status: asrTask.output.task_status,
|
||||
request_id: asrTask.request_id,
|
||||
mode: "async",
|
||||
model,
|
||||
};
|
||||
if (asrTask.output.results) base.results = asrTask.output.results;
|
||||
if (asrTask.output.result) {
|
||||
base.result = asrTask.output.result;
|
||||
if (typeof asrTask.output.result.transcription_url === "string") {
|
||||
base.transcription_url = asrTask.output.result.transcription_url;
|
||||
}
|
||||
}
|
||||
if (asrTask.output.task_metrics) base.task_metrics = asrTask.output.task_metrics;
|
||||
if (asrTask.usage) base.usage = asrTask.usage;
|
||||
|
||||
if (transcriptionItems.length === 0) {
|
||||
return base;
|
||||
}
|
||||
|
||||
const texts: string[] = [];
|
||||
const transcripts: Record<string, unknown>[] = [];
|
||||
for (const item of transcriptionItems) {
|
||||
if (!item.transcription_url) continue;
|
||||
const transRes = await fetch(item.transcription_url, { signal: ctx.signal });
|
||||
if (!transRes.ok) {
|
||||
throw new PipelineError(
|
||||
"async_task_failed",
|
||||
`Failed to download transcription: HTTP ${transRes.status}`,
|
||||
{ step: "speech/recognize", details: { taskId, url: item.transcription_url } },
|
||||
);
|
||||
}
|
||||
const transData = (await transRes.json()) as Record<string, unknown>;
|
||||
transcripts.push(transData);
|
||||
const transcriptList = transData.transcripts as
|
||||
| Array<{ text?: string; sentences?: Array<{ text?: string }> }>
|
||||
| undefined;
|
||||
if (!transcriptList?.length) continue;
|
||||
for (const transcript of transcriptList) {
|
||||
if (transcript.sentences?.length) {
|
||||
for (const sentence of transcript.sentences) {
|
||||
if (sentence.text) texts.push(sentence.text);
|
||||
}
|
||||
} else if (transcript.text) {
|
||||
texts.push(transcript.text);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
...base,
|
||||
text: texts.join("\n"),
|
||||
transcripts,
|
||||
};
|
||||
}
|
||||
|
||||
// --- Shared: task polling ---
|
||||
@@ -640,7 +779,7 @@ function flattenTaskResponse(resp: DashScopeTaskResponse): Record<string, unknow
|
||||
if (urls.length > 0) flat.urls = urls;
|
||||
}
|
||||
if (output.results) {
|
||||
const urls = output.results.map((r) => r.url).filter(Boolean);
|
||||
const urls = output.results.map((item) => item.url).filter(Boolean);
|
||||
if (urls.length > 0 && !flat.urls) flat.urls = urls;
|
||||
}
|
||||
if (output.task_metrics) flat.task_metrics = output.task_metrics;
|
||||
@@ -658,13 +797,13 @@ async function pollTask(
|
||||
return await pollTaskWithOptions(env, taskId, pollIntervalMs, timeoutMs, ctx);
|
||||
}
|
||||
|
||||
async function pollTaskWithOptions(
|
||||
async function pollUntilSucceeded(
|
||||
env: PipelineEnv,
|
||||
taskId: string,
|
||||
pollIntervalMs: number,
|
||||
timeoutMs: number,
|
||||
ctx?: StepContext,
|
||||
): Promise<Record<string, unknown>> {
|
||||
): Promise<DashScopeTaskResponse> {
|
||||
const started = Date.now();
|
||||
let attempt = 0;
|
||||
|
||||
@@ -687,7 +826,7 @@ async function pollTaskWithOptions(
|
||||
const status = result.output.task_status;
|
||||
|
||||
if (status === "SUCCEEDED") {
|
||||
return flattenTaskResponse(result);
|
||||
return result;
|
||||
}
|
||||
|
||||
if (status === "FAILED") {
|
||||
@@ -718,6 +857,33 @@ async function pollTaskWithOptions(
|
||||
}
|
||||
}
|
||||
|
||||
async function pollTaskWithOptions(
|
||||
env: PipelineEnv,
|
||||
taskId: string,
|
||||
pollIntervalMs: number,
|
||||
timeoutMs: number,
|
||||
ctx?: StepContext,
|
||||
): Promise<Record<string, unknown>> {
|
||||
return flattenTaskResponse(await pollUntilSucceeded(env, taskId, pollIntervalMs, timeoutMs, ctx));
|
||||
}
|
||||
|
||||
/** ASR task polling: preserve original output (includes results[] / result.transcription_url). */
|
||||
async function pollAsrTaskWithOptions(
|
||||
env: PipelineEnv,
|
||||
taskId: string,
|
||||
pollIntervalMs: number,
|
||||
timeoutMs: number,
|
||||
ctx?: StepContext,
|
||||
): Promise<DashScopeASRTaskResult> {
|
||||
return (await pollUntilSucceeded(
|
||||
env,
|
||||
taskId,
|
||||
pollIntervalMs,
|
||||
timeoutMs,
|
||||
ctx,
|
||||
)) as DashScopeASRTaskResult;
|
||||
}
|
||||
|
||||
function delay(ms: number, signal?: AbortSignal): Promise<void> {
|
||||
if (!signal) return new Promise((resolve) => setTimeout(resolve, ms));
|
||||
return new Promise((resolve, reject) => {
|
||||
|
||||
@@ -0,0 +1,184 @@
|
||||
import { expect, test } from "vite-plus/test";
|
||||
import type { Client } from "bailian-cli-core";
|
||||
import { PipelineError } from "../src/pipeline/errors.ts";
|
||||
import type { PipelineEnv } from "../src/pipeline/bl-config.ts";
|
||||
import { speechRecognize } from "../src/pipeline/steps/bl-api.ts";
|
||||
import type { StepContext } from "../src/pipeline/types.ts";
|
||||
|
||||
type CapturedRequest = {
|
||||
path?: string;
|
||||
method?: string;
|
||||
headers?: Record<string, string>;
|
||||
body?: Record<string, unknown>;
|
||||
async?: boolean;
|
||||
};
|
||||
|
||||
function makeEnv(requestJsonImpl?: (opts: CapturedRequest) => Promise<unknown>): {
|
||||
env: PipelineEnv;
|
||||
captured: CapturedRequest[];
|
||||
} {
|
||||
const captured: CapturedRequest[] = [];
|
||||
const client = {
|
||||
uploadFile: async (source: string) => source,
|
||||
requestJson: async (opts: CapturedRequest) => {
|
||||
captured.push(opts);
|
||||
if (requestJsonImpl) return requestJsonImpl(opts);
|
||||
return { output: { text: "ok" } };
|
||||
},
|
||||
} as unknown as Client;
|
||||
|
||||
return {
|
||||
env: {
|
||||
client,
|
||||
settings: { quiet: true, output: "json" } as PipelineEnv["settings"],
|
||||
},
|
||||
captured,
|
||||
};
|
||||
}
|
||||
|
||||
function makeCtx(): StepContext {
|
||||
return { dryRun: false, signal: new AbortController().signal };
|
||||
}
|
||||
|
||||
test("pipeline speechRecognize routes input-audio flash to sync multimodal endpoint", async () => {
|
||||
const { env, captured } = makeEnv();
|
||||
const result = (await speechRecognize(
|
||||
env,
|
||||
{
|
||||
url: "https://example.com/a.wav",
|
||||
model: "qwen-audio-3.0-asr-flash",
|
||||
language: "en",
|
||||
"vocabulary-id": "vocab-1",
|
||||
},
|
||||
makeCtx(),
|
||||
)) as { mode?: string; text?: string };
|
||||
|
||||
expect(result.mode).toBe("sync");
|
||||
expect(result.text).toBe("ok");
|
||||
expect(captured).toHaveLength(1);
|
||||
expect(captured[0]?.path).toBe("/api/v1/services/aigc/multimodal-generation/generation");
|
||||
expect(captured[0]?.headers?.["X-DashScope-SSE"]).toBe("disable");
|
||||
expect(captured[0]?.body).toMatchObject({
|
||||
model: "qwen-audio-3.0-asr-flash",
|
||||
parameters: {
|
||||
format: "wav",
|
||||
language_hints: ["en"],
|
||||
vocabulary_id: "vocab-1",
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
test("pipeline speechRecognize maps qwen3-filetrans language to parameters.language", async () => {
|
||||
const { env, captured } = makeEnv(async (opts) => {
|
||||
if (opts.async || opts.method === "POST") {
|
||||
return { output: { task_id: "task-1", task_status: "PENDING" } };
|
||||
}
|
||||
return {
|
||||
output: { task_id: "task-1", task_status: "SUCCEEDED", results: [] },
|
||||
request_id: "r1",
|
||||
};
|
||||
});
|
||||
|
||||
await speechRecognize(
|
||||
env,
|
||||
{
|
||||
url: "https://example.com/a.wav",
|
||||
model: "qwen3-asr-flash-filetrans",
|
||||
language: "zh",
|
||||
"poll-interval": 0,
|
||||
},
|
||||
makeCtx(),
|
||||
);
|
||||
|
||||
expect(captured[0]?.path).toBe("/api/v1/services/audio/asr/transcription");
|
||||
expect(captured[0]?.async).toBe(true);
|
||||
expect(captured[0]?.body).toMatchObject({
|
||||
model: "qwen3-asr-flash-filetrans",
|
||||
input: { file_url: "https://example.com/a.wav" },
|
||||
parameters: { language: "zh" },
|
||||
});
|
||||
expect(
|
||||
(captured[0]?.body?.parameters as Record<string, unknown> | undefined)?.language_hints,
|
||||
).toBeUndefined();
|
||||
});
|
||||
|
||||
test("pipeline speechRecognize rejects realtime models before requesting", async () => {
|
||||
const { env, captured } = makeEnv();
|
||||
await expect(
|
||||
speechRecognize(
|
||||
env,
|
||||
{ url: "https://example.com/a.wav", model: "qwen3-asr-flash-realtime" },
|
||||
makeCtx(),
|
||||
),
|
||||
).rejects.toBeInstanceOf(PipelineError);
|
||||
expect(captured).toHaveLength(0);
|
||||
});
|
||||
|
||||
test("pipeline speechRecognize rejects multiple urls for sync flash", async () => {
|
||||
const { env, captured } = makeEnv();
|
||||
await expect(
|
||||
speechRecognize(
|
||||
env,
|
||||
{
|
||||
url: ["https://example.com/a.wav", "https://example.com/b.wav"],
|
||||
model: "fun-asr-flash-2026-06-15",
|
||||
},
|
||||
makeCtx(),
|
||||
),
|
||||
).rejects.toBeInstanceOf(PipelineError);
|
||||
expect(captured).toHaveLength(0);
|
||||
});
|
||||
|
||||
test("pipeline speechRecognize downloads qwen3 singular result.transcription_url", async () => {
|
||||
const originalFetch = globalThis.fetch;
|
||||
const transcriptionUrl = "https://example.com/transcription.json";
|
||||
globalThis.fetch = (async (input: RequestInfo | URL) => {
|
||||
const url = typeof input === "string" ? input : input instanceof URL ? input.href : input.url;
|
||||
expect(url).toBe(transcriptionUrl);
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
transcripts: [{ text: "pipeline hello", sentences: [{ text: "pipeline hello" }] }],
|
||||
}),
|
||||
{ status: 200, headers: { "Content-Type": "application/json" } },
|
||||
);
|
||||
}) as typeof fetch;
|
||||
|
||||
try {
|
||||
const { env, captured } = makeEnv(async (opts) => {
|
||||
if (opts.async || opts.method === "POST") {
|
||||
return { output: { task_id: "task-1", task_status: "PENDING" } };
|
||||
}
|
||||
return {
|
||||
output: {
|
||||
task_id: "task-1",
|
||||
task_status: "SUCCEEDED",
|
||||
result: { transcription_url: transcriptionUrl },
|
||||
},
|
||||
request_id: "r1",
|
||||
};
|
||||
});
|
||||
|
||||
const result = (await speechRecognize(
|
||||
env,
|
||||
{
|
||||
url: "https://example.com/a.wav",
|
||||
model: "qwen3-asr-flash-filetrans",
|
||||
"poll-interval": 0,
|
||||
},
|
||||
makeCtx(),
|
||||
)) as {
|
||||
mode?: string;
|
||||
text?: string;
|
||||
transcription_url?: string;
|
||||
result?: { transcription_url?: string };
|
||||
};
|
||||
|
||||
expect(captured[0]?.async).toBe(true);
|
||||
expect(result.mode).toBe("async");
|
||||
expect(result.text).toBe("pipeline hello");
|
||||
expect(result.transcription_url).toBe(transcriptionUrl);
|
||||
expect(result.result?.transcription_url).toBe(transcriptionUrl);
|
||||
} finally {
|
||||
globalThis.fetch = originalFetch;
|
||||
}
|
||||
});
|
||||
@@ -39,6 +39,8 @@ description: >-
|
||||
| A/V understanding (files the host can't play) | `bl omni --video` / `--audio` | `qwen3.5-omni-plus` |
|
||||
| Image/video describe (user names Bailian) | `bl vision describe` | `qwen-vl-max`; host-first for plain image Q&A |
|
||||
|
||||
For ASR model selection, keep `fun-asr` (or other `*-filetrans`) for long recordings, repeated files, speaker diarization, or asynchronous task IDs. For one local or remote audio file up to about five minutes when the user asks for low-latency Flash models, use `--model fun-asr-flash-2026-06-15`, `--model qwen-audio-3.0-asr-flash`, or `--model qwen3-asr-flash`. Flash recognition is synchronous and accepts exactly one file per call.
|
||||
|
||||
Flags, usage, and examples: see [`reference/`](reference/index.md) or `bl <command> --help` — do not guess flags.
|
||||
|
||||
## Local files (mandatory)
|
||||
|
||||
@@ -14,7 +14,7 @@ Use this index for the skill-scoped quick index and global flags.
|
||||
| `bl image edit` | API Key | Edit an existing image with text instructions (Qwen-Image / Wan 2.7) | [image.md](image.md) |
|
||||
| `bl image generate` | API Key | Generate images (Qwen-Image / wan2.x) | [image.md](image.md) |
|
||||
| `bl omni` | API Key | Multimodal chat with text + audio output (Qwen-Omni) | [omni.md](omni.md) |
|
||||
| `bl speech recognize` | API Key | Recognize speech from audio files (FunAudio-ASR) | [speech.md](speech.md) |
|
||||
| `bl speech recognize` | API Key | Recognize speech from audio files (FunAudio-ASR / Qwen-ASR Flash) | [speech.md](speech.md) |
|
||||
| `bl speech synthesize` | API Key | Synthesize speech from text (CosyVoice TTS) | [speech.md](speech.md) |
|
||||
| `bl video download` | API Key | Download a completed video by task ID | [video.md](video.md) |
|
||||
| `bl video edit` | API Key | Edit a video with happyhorse-1.0-video-edit (style transfer, object replacement, etc.) | [video.md](video.md) |
|
||||
|
||||
@@ -7,38 +7,38 @@ Index: [index.md](index.md)
|
||||
|
||||
## Commands in this group
|
||||
|
||||
| Command | Authentication | Description |
|
||||
| ---------------------- | -------------- | ------------------------------------------------ |
|
||||
| `bl speech recognize` | API Key | Recognize speech from audio files (FunAudio-ASR) |
|
||||
| `bl speech synthesize` | API Key | Synthesize speech from text (CosyVoice TTS) |
|
||||
| Command | Authentication | Description |
|
||||
| ---------------------- | -------------- | ----------------------------------------------------------------- |
|
||||
| `bl speech recognize` | API Key | Recognize speech from audio files (FunAudio-ASR / Qwen-ASR Flash) |
|
||||
| `bl speech synthesize` | API Key | Synthesize speech from text (CosyVoice TTS) |
|
||||
|
||||
## Command details
|
||||
|
||||
### `bl speech recognize`
|
||||
|
||||
| Field | Value |
|
||||
| ------------------ | ------------------------------------------------ |
|
||||
| **Name** | `speech recognize` |
|
||||
| **Description** | Recognize speech from audio files (FunAudio-ASR) |
|
||||
| **Authentication** | API Key |
|
||||
| **Usage** | `bl speech recognize --url <audio-url> [flags]` |
|
||||
| Field | Value |
|
||||
| ------------------ | ----------------------------------------------------------------- |
|
||||
| **Name** | `speech recognize` |
|
||||
| **Description** | Recognize speech from audio files (FunAudio-ASR / Qwen-ASR Flash) |
|
||||
| **Authentication** | API Key |
|
||||
| **Usage** | `bl speech recognize --url <audio-url> [flags]` |
|
||||
|
||||
#### Flags
|
||||
|
||||
| Flag | Type | Required | Description |
|
||||
| --------------------------- | ------ | -------- | ------------------------------------------------------- |
|
||||
| `--url <url>` | array | yes | Audio file URL or local file path (repeatable, max 100) |
|
||||
| `--model <model>` | string | no | Model ID (default: fun-asr) |
|
||||
| `--language <lang>` | string | no | Language hint (e.g. zh, en, ja) |
|
||||
| `--diarization` | switch | no | Enable automatic speaker diarization |
|
||||
| `--speaker-count <n>` | number | no | Expected number of speakers (requires --diarization) |
|
||||
| `--vocabulary-id <id>` | string | no | Hot-word vocabulary ID for improved accuracy |
|
||||
| `--channel-id <n>` | number | no | Audio channel ID (default: 0) |
|
||||
| `--out <path>` | string | no | Save full transcription result to JSON file |
|
||||
| `--async` | switch | no | Return async task id without waiting |
|
||||
| `--poll-interval <seconds>` | number | no | Polling interval in seconds (default: 2) |
|
||||
| `--api-key <key>` | string | no | API key |
|
||||
| `--base-url <url>` | string | no | API base URL |
|
||||
| Flag | Type | Required | Description |
|
||||
| --------------------------- | ------ | -------- | ------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `--url <url>` | array | yes | Audio file URL or local file path (repeatable, max 100) |
|
||||
| `--model <model>` | string | no | Model ID (default: fun-asr). Async: fun-asr / _-filetrans / paraformer-_; sync: qwen3-asr-flash* / fun-asr-flash* / qwen-audio-\*-asr-flash |
|
||||
| `--language <lang>` | string | no | Language hint (e.g. zh, en, ja). Classic async/input-audio: language_hints; qwen3-filetrans: language; qwen3 sync: asr_options.language |
|
||||
| `--diarization` | switch | no | Enable automatic speaker diarization |
|
||||
| `--speaker-count <n>` | number | no | Expected number of speakers (requires --diarization) |
|
||||
| `--vocabulary-id <id>` | string | no | Hot-word vocabulary ID for improved accuracy |
|
||||
| `--channel-id <n>` | number | no | Audio channel ID (default: 0) |
|
||||
| `--out <path>` | string | no | Save full transcription result to JSON file |
|
||||
| `--async` | switch | no | Return async task id without waiting |
|
||||
| `--poll-interval <seconds>` | number | no | Polling interval in seconds (default: 2) |
|
||||
| `--api-key <key>` | string | no | API key |
|
||||
| `--base-url <url>` | string | no | API base URL |
|
||||
|
||||
#### Examples
|
||||
|
||||
@@ -70,6 +70,10 @@ bl speech recognize --url https://example.com/audio.mp3 --out result.json
|
||||
bl speech recognize --url https://example.com/audio.mp3 --async --quiet
|
||||
```
|
||||
|
||||
```bash
|
||||
bl speech recognize --url https://example.com/audio.mp3 --model qwen-audio-3.0-asr-flash --language en
|
||||
```
|
||||
|
||||
### `bl speech synthesize`
|
||||
|
||||
| Field | Value |
|
||||
|
||||
Reference in New Issue
Block a user