mirror of
https://github.com/modelstudioai/cli.git
synced 2026-09-14 19:49:23 +08:00
fix: support sync-flash and qwen3-filetrans ASR models in speech recognize
- Add asr-routes.ts with resolveAsrApi() to route models to the correct DashScope endpoint instead of always hitting asr/transcription - Async filetrans: fun-asr / paraformer / *-filetrans → file_urls (plural) - Async filetrans (qwen3): qwen3-asr-flash-filetrans* → file_url (singular) - Sync flash (input-audio): fun-asr-flash* / qwen-audio-*-asr-flash → multimodal-generation - Sync flash (qwen3): qwen3-asr-flash* → multimodal-generation + asr_options - Realtime/streaming models now give a clear USAGE error instead of a confusing server-side "url error" - Propagate same routing logic to pipeline speechRecognize step - Add table-driven unit tests and dry-run e2e assertions Fixes #146
This commit is contained in:
@@ -12,6 +12,11 @@ import {
|
||||
stripUndefined,
|
||||
taskPath,
|
||||
speechRecognizePath,
|
||||
resolveAsrApi,
|
||||
buildAsrFlashRequest,
|
||||
extractAsrFlashText,
|
||||
type AsrApiRoute,
|
||||
type AsrFlashFamily,
|
||||
type OutputFormat,
|
||||
type FlagsDef,
|
||||
type ParsedFlags,
|
||||
@@ -27,8 +32,18 @@ const RECOGNIZE_FLAGS = {
|
||||
description: "Audio file URL or local file path (repeatable, max 100)",
|
||||
required: true,
|
||||
},
|
||||
model: { type: "string", valueHint: "<model>", description: "Model ID (default: fun-asr)" },
|
||||
language: { type: "string", valueHint: "<lang>", description: "Language hint (e.g. zh, en, ja)" },
|
||||
model: {
|
||||
type: "string",
|
||||
valueHint: "<model>",
|
||||
description:
|
||||
"Model ID (default: fun-asr). Async: fun-asr / *-filetrans / paraformer-*; sync: qwen3-asr-flash* / fun-asr-flash* / qwen-audio-*-asr-flash",
|
||||
},
|
||||
language: {
|
||||
type: "string",
|
||||
valueHint: "<lang>",
|
||||
description:
|
||||
"Language hint (e.g. zh, en, ja). Async & input-audio sync: language_hints; qwen3 sync: asr_options.language",
|
||||
},
|
||||
diarization: { type: "switch", description: "Enable automatic speaker diarization" },
|
||||
speakerCount: {
|
||||
type: "number",
|
||||
@@ -55,8 +70,26 @@ const RECOGNIZE_FLAGS = {
|
||||
} satisfies FlagsDef;
|
||||
type RecognizeFlags = ParsedFlags<typeof RECOGNIZE_FLAGS>;
|
||||
|
||||
function assertSyncFlashFlagsAllowed(flags: RecognizeFlags, model: string): void {
|
||||
const unsupported: string[] = [];
|
||||
if (flags.diarization === true) unsupported.push("--diarization");
|
||||
if (flags.speakerCount !== undefined) unsupported.push("--speaker-count");
|
||||
if (flags.vocabularyId !== undefined) unsupported.push("--vocabulary-id");
|
||||
if (flags.channelId !== undefined) unsupported.push("--channel-id");
|
||||
if (flags.async === true) unsupported.push("--async");
|
||||
if (flags.pollInterval !== undefined) unsupported.push("--poll-interval");
|
||||
|
||||
if (unsupported.length > 0) {
|
||||
throw new BailianError(
|
||||
`Model "${model}" uses sync Flash ASR and does not support: ${unsupported.join(", ")}.\n` +
|
||||
`Hint: Use an async filetrans model (e.g. fun-asr, qwen3-asr-flash-filetrans) for those flags.`,
|
||||
ExitCode.USAGE,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
export default defineCommand({
|
||||
description: "Recognize speech from audio files (FunAudio-ASR)",
|
||||
description: "Recognize speech from audio files (FunAudio-ASR / Qwen-ASR Flash)",
|
||||
auth: "apiKey",
|
||||
usageArgs: "--url <audio-url> [flags]",
|
||||
flags: RECOGNIZE_FLAGS,
|
||||
@@ -68,6 +101,7 @@ export default defineCommand({
|
||||
"--url https://example.com/audio.mp3 --vocabulary-id vocab-abc123",
|
||||
"--url https://example.com/audio.mp3 --out result.json",
|
||||
"--url https://example.com/audio.mp3 --async --quiet",
|
||||
"--url https://example.com/audio.mp3 --model qwen-audio-3.0-asr-flash --language en",
|
||||
],
|
||||
async run(ctx) {
|
||||
const { settings, flags } = ctx;
|
||||
@@ -90,19 +124,64 @@ export default defineCommand({
|
||||
}
|
||||
|
||||
const model = flags.model || "fun-asr";
|
||||
const route = resolveAsrApi(model);
|
||||
if (route.kind === "unsupported") {
|
||||
throw new BailianError(
|
||||
route.unsupportedReason ?? `Unsupported ASR model: ${model}`,
|
||||
ExitCode.USAGE,
|
||||
);
|
||||
}
|
||||
|
||||
if (route.kind === "sync-flash") {
|
||||
assertSyncFlashFlagsAllowed(flags, model);
|
||||
if (rawUrls.length !== 1) {
|
||||
throw new BailianError(
|
||||
`Model "${model}" is a sync Flash ASR model and accepts exactly one --url (got ${rawUrls.length}).\n` +
|
||||
`Hint: Pass a single audio URL, or use an async filetrans model for batch files.`,
|
||||
ExitCode.USAGE,
|
||||
);
|
||||
}
|
||||
}
|
||||
if (
|
||||
route.kind === "async-filetrans" &&
|
||||
route.asyncInputStyle === "file_url" &&
|
||||
rawUrls.length !== 1
|
||||
) {
|
||||
throw new BailianError(
|
||||
`Model "${model}" accepts exactly one --url (got ${rawUrls.length}).\n` +
|
||||
"Hint: qwen3-asr-flash-filetrans* requires a single file_url.",
|
||||
ExitCode.USAGE,
|
||||
);
|
||||
}
|
||||
|
||||
const format = detectOutputFormat(settings.output);
|
||||
|
||||
// Auto-upload local files in parallel
|
||||
const resolvedUrls = await Promise.all(rawUrls.map((u) => ctx.client.uploadFile(u, model)));
|
||||
const resolvedUrls = await Promise.all(rawUrls.map((url) => ctx.client.uploadFile(url, model)));
|
||||
|
||||
if (route.kind === "sync-flash") {
|
||||
await handleSyncFlashMode(
|
||||
ctx.client,
|
||||
settings,
|
||||
flags,
|
||||
format,
|
||||
model,
|
||||
route,
|
||||
resolvedUrls[0]!,
|
||||
);
|
||||
return;
|
||||
}
|
||||
|
||||
const channelId = flags.channelId;
|
||||
const language = flags.language;
|
||||
const vocabularyId = flags.vocabularyId;
|
||||
|
||||
const body: DashScopeASRRequest = {
|
||||
model,
|
||||
input: {
|
||||
file_urls: resolvedUrls,
|
||||
},
|
||||
input:
|
||||
route.asyncInputStyle === "file_url"
|
||||
? { file_url: resolvedUrls[0]! }
|
||||
: { file_urls: resolvedUrls },
|
||||
parameters: {
|
||||
channel_id: channelId !== undefined ? [channelId] : [0],
|
||||
language_hints: language ? [language] : undefined,
|
||||
@@ -116,7 +195,7 @@ export default defineCommand({
|
||||
stripUndefined(body.parameters as Record<string, unknown>);
|
||||
|
||||
if (settings.dryRun) {
|
||||
emitResult({ request: body, mode: "async" }, format);
|
||||
emitResult({ request: body, mode: "async", path: speechRecognizePath() }, format);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -128,6 +207,53 @@ export default defineCommand({
|
||||
},
|
||||
});
|
||||
|
||||
async function handleSyncFlashMode(
|
||||
client: Client,
|
||||
settings: Settings,
|
||||
flags: RecognizeFlags,
|
||||
format: OutputFormat,
|
||||
model: string,
|
||||
route: AsrApiRoute,
|
||||
audioUrl: string,
|
||||
): Promise<void> {
|
||||
const flashFamily = route.flashFamily as AsrFlashFamily;
|
||||
const body = buildAsrFlashRequest({
|
||||
model,
|
||||
audioUrl,
|
||||
language: flags.language,
|
||||
flashFamily,
|
||||
});
|
||||
|
||||
if (settings.dryRun) {
|
||||
emitResult({ request: body, mode: "sync", path: route.path }, format);
|
||||
return;
|
||||
}
|
||||
|
||||
if (!settings.quiet) {
|
||||
process.stderr.write(`[Model: ${model}] [Mode: sync] [Files: 1]\n`);
|
||||
}
|
||||
|
||||
const response = await client.requestJson<Record<string, unknown>>({
|
||||
path: route.path,
|
||||
method: "POST",
|
||||
body,
|
||||
});
|
||||
|
||||
const text = extractAsrFlashText(response, flashFamily);
|
||||
if (text) {
|
||||
process.stdout.write(text.endsWith("\n") ? text : `${text}\n`);
|
||||
} else {
|
||||
emitBare(JSON.stringify(response));
|
||||
}
|
||||
|
||||
if (flags.out) {
|
||||
writeFileSync(flags.out, JSON.stringify(response, null, 2) + "\n");
|
||||
if (!settings.quiet) {
|
||||
process.stderr.write(`Full result saved to: ${flags.out}\n`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async function handleAsyncMode(
|
||||
client: Client,
|
||||
settings: Settings,
|
||||
@@ -160,12 +286,12 @@ async function handleAsyncMode(
|
||||
url: pollUrl,
|
||||
intervalSec: pollInterval,
|
||||
timeoutSec: settings.timeout,
|
||||
isComplete: (d) => (d as DashScopeASRTaskResult).output.task_status === "SUCCEEDED",
|
||||
isFailed: (d) => (d as DashScopeASRTaskResult).output.task_status === "FAILED",
|
||||
getStatus: (d) => (d as DashScopeASRTaskResult).output.task_status,
|
||||
getErrorMessage: (d) => {
|
||||
const o = (d as DashScopeASRTaskResult).output;
|
||||
return (o as unknown as Record<string, unknown>).message as string | undefined;
|
||||
isComplete: (data) => (data as DashScopeASRTaskResult).output.task_status === "SUCCEEDED",
|
||||
isFailed: (data) => (data as DashScopeASRTaskResult).output.task_status === "FAILED",
|
||||
getStatus: (data) => (data as DashScopeASRTaskResult).output.task_status,
|
||||
getErrorMessage: (data) => {
|
||||
const output = (data as DashScopeASRTaskResult).output;
|
||||
return (output as unknown as Record<string, unknown>).message as string | undefined;
|
||||
},
|
||||
});
|
||||
|
||||
@@ -179,12 +305,14 @@ async function handleAsyncMode(
|
||||
// Collect all transcription data for --out
|
||||
const allTransData: Record<string, unknown>[] = [];
|
||||
|
||||
for (let i = 0; i < results.length; i++) {
|
||||
const subResult = results[i]!;
|
||||
for (let index = 0; index < results.length; index++) {
|
||||
const subResult = results[index]!;
|
||||
const isMulti = fileCount > 1;
|
||||
|
||||
if (isMulti) {
|
||||
process.stdout.write(`=== [${i + 1}/${results.length}] ${subResult.file_url ?? ""} ===\n`);
|
||||
process.stdout.write(
|
||||
`=== [${index + 1}/${results.length}] ${subResult.file_url ?? ""} ===\n`,
|
||||
);
|
||||
}
|
||||
|
||||
if (subResult.subtask_status === "FAILED") {
|
||||
|
||||
@@ -16,6 +16,32 @@ import { SPEECH_ROUTES } from "./topic-routes.ts";
|
||||
*/
|
||||
|
||||
describe("e2e: speech recognize", () => {
|
||||
async function runRecognizeDryRun(args: string[]) {
|
||||
const { stdout, stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [
|
||||
"speech",
|
||||
"recognize",
|
||||
...args,
|
||||
"--dry-run",
|
||||
"--output",
|
||||
"json",
|
||||
"--quiet",
|
||||
]);
|
||||
expect(exitCode, stderr).toBe(0);
|
||||
return parseStdoutJson<{
|
||||
mode?: string;
|
||||
path?: string;
|
||||
request?: {
|
||||
model?: string;
|
||||
parameters?: { format?: string; language_hints?: string[] };
|
||||
input?: {
|
||||
file_url?: string;
|
||||
file_urls?: string[];
|
||||
messages?: Array<{ content?: Array<{ type?: string }> }>;
|
||||
};
|
||||
};
|
||||
}>(stdout);
|
||||
}
|
||||
|
||||
test("speech recognize --help 正常退出", async () => {
|
||||
const { stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [
|
||||
"speech",
|
||||
@@ -25,6 +51,50 @@ describe("e2e: speech recognize", () => {
|
||||
expect(exitCode, stderr).toBe(0);
|
||||
expect(stderr).toMatch(/recognize|--url|model|audio/i);
|
||||
});
|
||||
|
||||
test("speech recognize sync-flash dry-run 走 multimodal-generation", async () => {
|
||||
const body = await runRecognizeDryRun([
|
||||
"--model",
|
||||
"qwen-audio-3.0-asr-flash",
|
||||
"--url",
|
||||
"https://dashscope.oss-cn-beijing.aliyuncs.com/samples/audio/paraformer/hello_world_female2.wav",
|
||||
"--language",
|
||||
"en",
|
||||
]);
|
||||
expect(body.mode).toBe("sync");
|
||||
expect(body.path).toBe("/api/v1/services/aigc/multimodal-generation/generation");
|
||||
expect(body.request?.model).toBe("qwen-audio-3.0-asr-flash");
|
||||
expect(body.request?.parameters?.format).toBe("wav");
|
||||
expect(body.request?.parameters?.language_hints).toEqual(["en"]);
|
||||
expect(body.request?.input?.messages?.[0]?.content?.[0]?.type).toBe("input_audio");
|
||||
});
|
||||
|
||||
test("speech recognize qwen3 filetrans dry-run 使用 file_url 单数字段", async () => {
|
||||
const body = await runRecognizeDryRun([
|
||||
"--model",
|
||||
"qwen3-asr-flash-filetrans",
|
||||
"--url",
|
||||
"https://dashscope.oss-cn-beijing.aliyuncs.com/samples/audio/paraformer/hello_world_female2.wav",
|
||||
]);
|
||||
expect(body.mode).toBe("async");
|
||||
expect(body.path).toBe("/api/v1/services/audio/asr/transcription");
|
||||
expect(body.request?.input?.file_url?.startsWith("https://")).toBe(true);
|
||||
expect(body.request?.input?.file_urls).toBeUndefined();
|
||||
});
|
||||
|
||||
test("speech recognize realtime 模型报用法错误", async () => {
|
||||
const { stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [
|
||||
"speech",
|
||||
"recognize",
|
||||
"--model",
|
||||
"qwen3-asr-flash-realtime",
|
||||
"--url",
|
||||
"https://example.com/a.wav",
|
||||
"--quiet",
|
||||
]);
|
||||
expect(exitCode).toBe(2);
|
||||
expect(stderr).toMatch(/realtime|WebSocket|unsupported/i);
|
||||
});
|
||||
});
|
||||
|
||||
describe.skipIf(!isBailianE2EMediaEnabled() || !isDashScopeE2EReady())(
|
||||
|
||||
@@ -0,0 +1,254 @@
|
||||
import { imageSyncPath, speechRecognizePath } from "./endpoints.ts";
|
||||
|
||||
/**
|
||||
* DashScope ASR APIs differ by model family:
|
||||
*
|
||||
* - async file transcription (`.../audio/asr/transcription`):
|
||||
* fun-asr*, paraformer* (non-realtime), *-filetrans, sensevoice*
|
||||
* language via `parameters.language_hints`
|
||||
* - sync multimodal (`.../aigc/multimodal-generation/generation`):
|
||||
* - qwen3: `{ content: [{ audio }] }` + optional `asr_options.language`
|
||||
* (qwen3-asr-flash*)
|
||||
* - input-audio: `{ type: input_audio, input_audio.data }` +
|
||||
* `format`/`sample_rate` + optional `language_hints`
|
||||
* (fun-asr-flash*, qwen-audio-*-asr-flash*)
|
||||
* - realtime / streaming: WebSocket — not supported by `speech recognize`
|
||||
*/
|
||||
|
||||
export type AsrApiKind = "async-filetrans" | "sync-flash" | "unsupported";
|
||||
|
||||
/** Sync-flash request body shape differs by Flash protocol family. */
|
||||
export type AsrFlashFamily = "qwen3" | "input-audio";
|
||||
|
||||
export interface AsrApiRoute {
|
||||
kind: AsrApiKind;
|
||||
path: string;
|
||||
/** True when the call is synchronous (no X-DashScope-Async / task poll). */
|
||||
useSync: boolean;
|
||||
/**
|
||||
* Async transcription request input style.
|
||||
* - `file_urls`: classic async models (fun-asr / paraformer / qwen-audio filetrans...)
|
||||
* - `file_url`: qwen3-asr-flash-filetrans family
|
||||
*/
|
||||
asyncInputStyle?: "file_urls" | "file_url";
|
||||
flashFamily?: AsrFlashFamily;
|
||||
/** Human-readable reason when kind is unsupported. */
|
||||
unsupportedReason?: string;
|
||||
}
|
||||
|
||||
function isRealtimeOrStreaming(model: string): boolean {
|
||||
return /realtime|streaming/i.test(model);
|
||||
}
|
||||
|
||||
function isFiletransModel(model: string): boolean {
|
||||
return /filetrans/i.test(model);
|
||||
}
|
||||
|
||||
function isQwen3FiletransModel(model: string): boolean {
|
||||
return /^qwen3-asr-flash-filetrans(?:-|$)/i.test(model);
|
||||
}
|
||||
|
||||
const INPUT_AUDIO_FLASH_PREFIXES = ["fun-asr-flash", "qwen-audio"] as const;
|
||||
|
||||
/**
|
||||
* Fun-ASR-Flash / Qwen-Audio-*-ASR-Flash share the input_audio + format protocol.
|
||||
* Examples: fun-asr-flash-2026-06-15, qwen-audio-3.0-asr-flash
|
||||
*/
|
||||
function isInputAudioFlashModel(model: string): boolean {
|
||||
if (isRealtimeOrStreaming(model) || isFiletransModel(model)) return false;
|
||||
if (model.startsWith(INPUT_AUDIO_FLASH_PREFIXES[0])) return true;
|
||||
if (model.startsWith(INPUT_AUDIO_FLASH_PREFIXES[1]) && /asr-flash/i.test(model)) return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Qwen3-ASR-Flash sync models use content.audio + asr_options.
|
||||
* Examples: qwen3-asr-flash, qwen3-asr-flash-2025-09-08, qwen3-asr-flash-us
|
||||
*/
|
||||
function isQwen3AsrFlashModel(model: string): boolean {
|
||||
if (!/^qwen3-asr-flash(?:-|$)/i.test(model)) return false;
|
||||
if (isFiletransModel(model) || isRealtimeOrStreaming(model)) return false;
|
||||
if (isInputAudioFlashModel(model)) return false;
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve which DashScope ASR API a model should use for file recognition.
|
||||
* Unknown models default to async-filetrans (preserves existing CLI behavior).
|
||||
*/
|
||||
export function resolveAsrApi(model: string): AsrApiRoute {
|
||||
if (isRealtimeOrStreaming(model)) {
|
||||
return {
|
||||
kind: "unsupported",
|
||||
path: "",
|
||||
useSync: false,
|
||||
unsupportedReason:
|
||||
`Model "${model}" is a realtime/streaming ASR model and requires a WebSocket API. ` +
|
||||
`Use an async filetrans model (e.g. fun-asr, qwen3-asr-flash-filetrans) or a sync flash model ` +
|
||||
`(e.g. qwen3-asr-flash, qwen-audio-3.0-asr-flash) with this command.`,
|
||||
};
|
||||
}
|
||||
|
||||
if (isFiletransModel(model)) {
|
||||
return {
|
||||
kind: "async-filetrans",
|
||||
path: speechRecognizePath(),
|
||||
useSync: false,
|
||||
asyncInputStyle: isQwen3FiletransModel(model) ? "file_url" : "file_urls",
|
||||
};
|
||||
}
|
||||
|
||||
if (isInputAudioFlashModel(model)) {
|
||||
return {
|
||||
kind: "sync-flash",
|
||||
path: imageSyncPath(),
|
||||
useSync: true,
|
||||
flashFamily: "input-audio",
|
||||
};
|
||||
}
|
||||
|
||||
if (isQwen3AsrFlashModel(model)) {
|
||||
return {
|
||||
kind: "sync-flash",
|
||||
path: imageSyncPath(),
|
||||
useSync: true,
|
||||
flashFamily: "qwen3",
|
||||
};
|
||||
}
|
||||
|
||||
// fun-asr / paraformer / sensevoice / unknown → keep legacy async path
|
||||
return {
|
||||
kind: "async-filetrans",
|
||||
path: speechRecognizePath(),
|
||||
useSync: false,
|
||||
asyncInputStyle: "file_urls",
|
||||
};
|
||||
}
|
||||
|
||||
/** Infer audio container hint for input-audio Flash `parameters.format`. */
|
||||
export function inferAudioFormatHint(audioUrl: string): string {
|
||||
const pathPart = audioUrl.split("?")[0] ?? audioUrl;
|
||||
const match = pathPart.match(/\.([a-zA-Z0-9]+)$/);
|
||||
const extension = match?.[1]?.toLowerCase();
|
||||
if (!extension) return "wav";
|
||||
if (extension === "mpeg") return "mp3";
|
||||
return extension;
|
||||
}
|
||||
|
||||
export interface BuildAsrFlashRequestOpts {
|
||||
model: string;
|
||||
audioUrl: string;
|
||||
language?: string;
|
||||
flashFamily: AsrFlashFamily;
|
||||
}
|
||||
|
||||
/** Build a sync multimodal ASR request body for Flash models. */
|
||||
export function buildAsrFlashRequest(opts: BuildAsrFlashRequestOpts): Record<string, unknown> {
|
||||
const { model, audioUrl, language, flashFamily } = opts;
|
||||
|
||||
if (flashFamily === "input-audio") {
|
||||
// 与官方 Qwen-Audio / Fun-ASR-Flash 文档一致:语种走 language_hints
|
||||
const parameters: Record<string, unknown> = {
|
||||
format: inferAudioFormatHint(audioUrl),
|
||||
sample_rate: "16000",
|
||||
};
|
||||
if (language) {
|
||||
parameters.language_hints = [language];
|
||||
}
|
||||
return {
|
||||
model,
|
||||
input: {
|
||||
messages: [
|
||||
{
|
||||
role: "user",
|
||||
content: [
|
||||
{
|
||||
type: "input_audio",
|
||||
input_audio: { data: audioUrl },
|
||||
},
|
||||
],
|
||||
},
|
||||
],
|
||||
},
|
||||
parameters,
|
||||
};
|
||||
}
|
||||
|
||||
const asrOptions: Record<string, unknown> = {};
|
||||
if (language) {
|
||||
asrOptions.language = language;
|
||||
}
|
||||
|
||||
const parameters: Record<string, unknown> = {};
|
||||
if (Object.keys(asrOptions).length > 0) {
|
||||
parameters.asr_options = asrOptions;
|
||||
}
|
||||
|
||||
const body: Record<string, unknown> = {
|
||||
model,
|
||||
input: {
|
||||
messages: [
|
||||
{
|
||||
role: "user",
|
||||
content: [{ audio: audioUrl }],
|
||||
},
|
||||
],
|
||||
},
|
||||
};
|
||||
if (Object.keys(parameters).length > 0) {
|
||||
body.parameters = parameters;
|
||||
}
|
||||
return body;
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract recognition text from a sync Flash ASR response.
|
||||
* Qwen3 uses choices[].message.content; input-audio Flash uses output.text.
|
||||
*/
|
||||
export function extractAsrFlashText(
|
||||
response: Record<string, unknown>,
|
||||
flashFamily: AsrFlashFamily,
|
||||
): string {
|
||||
const output = response.output as Record<string, unknown> | undefined;
|
||||
if (!output) return "";
|
||||
|
||||
if (flashFamily === "input-audio") {
|
||||
if (typeof output.text === "string" && output.text.length > 0) {
|
||||
return output.text;
|
||||
}
|
||||
const nested = output.output as Record<string, unknown> | undefined;
|
||||
const sentence = nested?.sentence as Record<string, unknown> | undefined;
|
||||
if (typeof sentence?.text === "string") {
|
||||
return sentence.text;
|
||||
}
|
||||
return "";
|
||||
}
|
||||
|
||||
const choices = output.choices as Array<Record<string, unknown>> | undefined;
|
||||
if (!choices?.length) return "";
|
||||
|
||||
const texts: string[] = [];
|
||||
for (const choice of choices) {
|
||||
const message = choice.message as Record<string, unknown> | undefined;
|
||||
if (!message) continue;
|
||||
const content = message.content;
|
||||
if (typeof content === "string") {
|
||||
texts.push(content);
|
||||
continue;
|
||||
}
|
||||
if (!Array.isArray(content)) continue;
|
||||
for (const item of content) {
|
||||
if (typeof item === "string") {
|
||||
texts.push(item);
|
||||
continue;
|
||||
}
|
||||
if (item && typeof item === "object") {
|
||||
const record = item as Record<string, unknown>;
|
||||
if (typeof record.text === "string") {
|
||||
texts.push(record.text);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return texts.join("");
|
||||
}
|
||||
@@ -34,6 +34,16 @@ export {
|
||||
type ImageInputStyle,
|
||||
type ImageSizeProfile,
|
||||
} from "./image-routes.ts";
|
||||
export {
|
||||
buildAsrFlashRequest,
|
||||
extractAsrFlashText,
|
||||
inferAudioFormatHint,
|
||||
resolveAsrApi,
|
||||
type AsrApiKind,
|
||||
type AsrApiRoute,
|
||||
type AsrFlashFamily,
|
||||
type BuildAsrFlashRequestOpts,
|
||||
} from "./asr-routes.ts";
|
||||
export { CHANNEL, sourceConfig, trackingHeaders, type TrackingIdentity } from "./headers.ts";
|
||||
export type { HttpDeps, RequestOpts } from "./http.ts";
|
||||
export { request, requestJson } from "./http.ts";
|
||||
|
||||
@@ -533,7 +533,8 @@ export interface DashScopeTTSStreamChunk {
|
||||
export interface DashScopeASRRequest {
|
||||
model: string;
|
||||
input: {
|
||||
file_urls: string[];
|
||||
file_urls?: string[];
|
||||
file_url?: string;
|
||||
};
|
||||
parameters?: {
|
||||
channel_id?: number[];
|
||||
|
||||
@@ -0,0 +1,154 @@
|
||||
import { expect, test } from "vite-plus/test";
|
||||
import {
|
||||
buildAsrFlashRequest,
|
||||
extractAsrFlashText,
|
||||
inferAudioFormatHint,
|
||||
resolveAsrApi,
|
||||
} from "../src/client/asr-routes.ts";
|
||||
|
||||
test("resolveAsrApi routes model families correctly", () => {
|
||||
const cases = [
|
||||
{
|
||||
model: "fun-asr",
|
||||
expected: {
|
||||
kind: "async-filetrans",
|
||||
useSync: false,
|
||||
path: "/api/v1/services/audio/asr/transcription",
|
||||
asyncInputStyle: "file_urls",
|
||||
},
|
||||
},
|
||||
{
|
||||
model: "qwen3-asr-flash-filetrans-2025-11-17",
|
||||
expected: {
|
||||
kind: "async-filetrans",
|
||||
useSync: false,
|
||||
path: "/api/v1/services/audio/asr/transcription",
|
||||
asyncInputStyle: "file_url",
|
||||
},
|
||||
},
|
||||
{
|
||||
model: "qwen-audio-3.0-asr-flash-filetrans",
|
||||
expected: {
|
||||
kind: "async-filetrans",
|
||||
useSync: false,
|
||||
asyncInputStyle: "file_urls",
|
||||
},
|
||||
},
|
||||
{
|
||||
model: "qwen3-asr-flash-us",
|
||||
expected: {
|
||||
kind: "sync-flash",
|
||||
useSync: true,
|
||||
flashFamily: "qwen3",
|
||||
path: "/api/v1/services/aigc/multimodal-generation/generation",
|
||||
},
|
||||
},
|
||||
{
|
||||
model: "qwen-audio-3.0-asr-flash",
|
||||
expected: {
|
||||
kind: "sync-flash",
|
||||
useSync: true,
|
||||
flashFamily: "input-audio",
|
||||
},
|
||||
},
|
||||
{
|
||||
model: "qwen3-asr-flash-realtime",
|
||||
expected: {
|
||||
kind: "unsupported",
|
||||
},
|
||||
},
|
||||
{
|
||||
model: "foo-asr-flash",
|
||||
expected: {
|
||||
kind: "async-filetrans",
|
||||
useSync: false,
|
||||
path: "/api/v1/services/audio/asr/transcription",
|
||||
asyncInputStyle: "file_urls",
|
||||
},
|
||||
},
|
||||
] as const;
|
||||
|
||||
for (const { model, expected } of cases) {
|
||||
const route = resolveAsrApi(model);
|
||||
expect(route, model).toMatchObject(expected);
|
||||
if (expected.kind === "unsupported") {
|
||||
expect(route.unsupportedReason, model).toMatch(/realtime|streaming|WebSocket/i);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
test("unknown models default to async-filetrans for backward compatibility", () => {
|
||||
expect(resolveAsrApi("custom-asr-model")).toMatchObject({
|
||||
kind: "async-filetrans",
|
||||
useSync: false,
|
||||
});
|
||||
});
|
||||
|
||||
test("inferAudioFormatHint reads extension from url", () => {
|
||||
expect(inferAudioFormatHint("https://example.com/a.mp3")).toBe("mp3");
|
||||
expect(inferAudioFormatHint("oss://bucket/path/file.WAV")).toBe("wav");
|
||||
expect(inferAudioFormatHint("https://example.com/a.mpeg?x=1")).toBe("mp3");
|
||||
expect(inferAudioFormatHint("https://example.com/noext")).toBe("wav");
|
||||
});
|
||||
|
||||
test("buildAsrFlashRequest shapes qwen3 and input-audio bodies", () => {
|
||||
expect(
|
||||
buildAsrFlashRequest({
|
||||
model: "qwen3-asr-flash",
|
||||
audioUrl: "https://example.com/a.mp3",
|
||||
language: "en",
|
||||
flashFamily: "qwen3",
|
||||
}),
|
||||
).toEqual({
|
||||
model: "qwen3-asr-flash",
|
||||
input: {
|
||||
messages: [{ role: "user", content: [{ audio: "https://example.com/a.mp3" }] }],
|
||||
},
|
||||
parameters: { asr_options: { language: "en" } },
|
||||
});
|
||||
|
||||
expect(
|
||||
buildAsrFlashRequest({
|
||||
model: "qwen-audio-3.0-asr-flash",
|
||||
audioUrl: "https://example.com/a.wav",
|
||||
language: "en",
|
||||
flashFamily: "input-audio",
|
||||
}),
|
||||
).toEqual({
|
||||
model: "qwen-audio-3.0-asr-flash",
|
||||
input: {
|
||||
messages: [
|
||||
{
|
||||
role: "user",
|
||||
content: [{ type: "input_audio", input_audio: { data: "https://example.com/a.wav" } }],
|
||||
},
|
||||
],
|
||||
},
|
||||
parameters: { format: "wav", sample_rate: "16000", language_hints: ["en"] },
|
||||
});
|
||||
});
|
||||
|
||||
test("extractAsrFlashText reads qwen3 choices and input-audio text fields", () => {
|
||||
expect(
|
||||
extractAsrFlashText(
|
||||
{
|
||||
output: {
|
||||
choices: [{ message: { content: [{ text: "你好" }] } }],
|
||||
},
|
||||
},
|
||||
"qwen3",
|
||||
),
|
||||
).toBe("你好");
|
||||
|
||||
expect(
|
||||
extractAsrFlashText(
|
||||
{
|
||||
output: {
|
||||
text: "Hello World",
|
||||
output: { sentence: { text: "ignored when text present" } },
|
||||
},
|
||||
},
|
||||
"input-audio",
|
||||
),
|
||||
).toBe("Hello World");
|
||||
});
|
||||
@@ -10,6 +10,9 @@ import {
|
||||
taskPath,
|
||||
speechSynthesizePath,
|
||||
speechRecognizePath,
|
||||
resolveAsrApi,
|
||||
buildAsrFlashRequest,
|
||||
extractAsrFlashText,
|
||||
stripUndefined,
|
||||
resolveBooleanFlag,
|
||||
resolveWatermark,
|
||||
@@ -573,24 +576,91 @@ export async function speechRecognize(
|
||||
});
|
||||
}
|
||||
|
||||
const model = input.model || "fun-asr";
|
||||
const route = resolveAsrApi(model);
|
||||
if (route.kind === "unsupported") {
|
||||
throw new PipelineError(
|
||||
"invalid_input",
|
||||
route.unsupportedReason ?? `Unsupported ASR model: ${model}`,
|
||||
{
|
||||
step: "speech/recognize",
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
if (route.kind === "sync-flash") {
|
||||
if (rawUrls.length !== 1) {
|
||||
throw new PipelineError(
|
||||
"invalid_input",
|
||||
`Model "${model}" is a sync Flash ASR model and accepts exactly one url (got ${rawUrls.length})`,
|
||||
{ step: "speech/recognize" },
|
||||
);
|
||||
}
|
||||
if (
|
||||
input.diarization ||
|
||||
input["speaker-count"] !== undefined ||
|
||||
input["vocabulary-id"] !== undefined ||
|
||||
input["channel-id"] !== undefined
|
||||
) {
|
||||
throw new PipelineError(
|
||||
"invalid_input",
|
||||
`Model "${model}" uses sync Flash ASR and does not support diarization / speaker-count / vocabulary-id / channel-id`,
|
||||
{ step: "speech/recognize" },
|
||||
);
|
||||
}
|
||||
}
|
||||
if (
|
||||
route.kind === "async-filetrans" &&
|
||||
route.asyncInputStyle === "file_url" &&
|
||||
rawUrls.length !== 1
|
||||
) {
|
||||
throw new PipelineError(
|
||||
"invalid_input",
|
||||
`Model "${model}" accepts exactly one url (got ${rawUrls.length})`,
|
||||
{ step: "speech/recognize" },
|
||||
);
|
||||
}
|
||||
|
||||
// Resolve local files to upload URLs
|
||||
const fileUrls: string[] = [];
|
||||
for (const u of rawUrls) {
|
||||
if (isLocalFile(u)) {
|
||||
for (const audioUrl of rawUrls) {
|
||||
if (isLocalFile(audioUrl)) {
|
||||
fileUrls.push(
|
||||
await env.client.uploadFile(u, input.model || "fun-asr", {
|
||||
await env.client.uploadFile(audioUrl, model, {
|
||||
signal: ctx.signal,
|
||||
}),
|
||||
);
|
||||
} else {
|
||||
fileUrls.push(u);
|
||||
fileUrls.push(audioUrl);
|
||||
}
|
||||
}
|
||||
|
||||
const model = input.model || "fun-asr";
|
||||
if (route.kind === "sync-flash") {
|
||||
const flashFamily = route.flashFamily!;
|
||||
const body = buildAsrFlashRequest({
|
||||
model,
|
||||
audioUrl: fileUrls[0]!,
|
||||
language: input.language,
|
||||
flashFamily,
|
||||
});
|
||||
const response = await env.client.requestJson<Record<string, unknown>>({
|
||||
path: route.path,
|
||||
method: "POST",
|
||||
body,
|
||||
signal: ctx.signal,
|
||||
});
|
||||
return {
|
||||
text: extractAsrFlashText(response, flashFamily),
|
||||
model,
|
||||
mode: "sync",
|
||||
raw: response,
|
||||
};
|
||||
}
|
||||
|
||||
const body: DashScopeASRRequest = {
|
||||
model,
|
||||
input: { file_urls: fileUrls },
|
||||
input:
|
||||
route.asyncInputStyle === "file_url" ? { file_url: fileUrls[0]! } : { file_urls: fileUrls },
|
||||
parameters: {
|
||||
channel_id: input["channel-id"] !== undefined ? [input["channel-id"]] : undefined,
|
||||
language_hints: input.language ? [input.language] : undefined,
|
||||
@@ -601,9 +671,8 @@ export async function speechRecognize(
|
||||
};
|
||||
stripUndefined(body.parameters as Record<string, unknown>);
|
||||
|
||||
const url = speechRecognizePath();
|
||||
const asyncResp = await env.client.requestJson<DashScopeAsyncResponse>({
|
||||
path: url,
|
||||
path: speechRecognizePath(),
|
||||
method: "POST",
|
||||
body,
|
||||
async: true,
|
||||
|
||||
@@ -14,7 +14,7 @@ Use this index for the skill-scoped quick index and global flags.
|
||||
| `bl image edit` | Edit an existing image with text instructions (Qwen-Image / Wan 2.7) | [image.md](image.md) |
|
||||
| `bl image generate` | Generate images (Qwen-Image / wan2.x) | [image.md](image.md) |
|
||||
| `bl omni` | Multimodal chat with text + audio output (Qwen-Omni) | [omni.md](omni.md) |
|
||||
| `bl speech recognize` | Recognize speech from audio files (FunAudio-ASR) | [speech.md](speech.md) |
|
||||
| `bl speech recognize` | Recognize speech from audio files (FunAudio-ASR / Qwen-ASR Flash) | [speech.md](speech.md) |
|
||||
| `bl speech synthesize` | Synthesize speech from text (CosyVoice TTS) | [speech.md](speech.md) |
|
||||
| `bl video download` | Download a completed video by task ID | [video.md](video.md) |
|
||||
| `bl video edit` | Edit a video with happyhorse-1.0-video-edit (style transfer, object replacement, etc.) | [video.md](video.md) |
|
||||
|
||||
@@ -7,37 +7,37 @@ Index: [index.md](index.md)
|
||||
|
||||
## Commands in this group
|
||||
|
||||
| Command | Description |
|
||||
| ---------------------- | ------------------------------------------------ |
|
||||
| `bl speech recognize` | Recognize speech from audio files (FunAudio-ASR) |
|
||||
| `bl speech synthesize` | Synthesize speech from text (CosyVoice TTS) |
|
||||
| Command | Description |
|
||||
| ---------------------- | ----------------------------------------------------------------- |
|
||||
| `bl speech recognize` | Recognize speech from audio files (FunAudio-ASR / Qwen-ASR Flash) |
|
||||
| `bl speech synthesize` | Synthesize speech from text (CosyVoice TTS) |
|
||||
|
||||
## Command details
|
||||
|
||||
### `bl speech recognize`
|
||||
|
||||
| Field | Value |
|
||||
| --------------- | ------------------------------------------------ |
|
||||
| **Name** | `speech recognize` |
|
||||
| **Description** | Recognize speech from audio files (FunAudio-ASR) |
|
||||
| **Usage** | `bl speech recognize --url <audio-url> [flags]` |
|
||||
| Field | Value |
|
||||
| --------------- | ----------------------------------------------------------------- |
|
||||
| **Name** | `speech recognize` |
|
||||
| **Description** | Recognize speech from audio files (FunAudio-ASR / Qwen-ASR Flash) |
|
||||
| **Usage** | `bl speech recognize --url <audio-url> [flags]` |
|
||||
|
||||
#### Flags
|
||||
|
||||
| Flag | Type | Required | Description |
|
||||
| --------------------------- | ------ | -------- | ------------------------------------------------------- |
|
||||
| `--url <url>` | array | yes | Audio file URL or local file path (repeatable, max 100) |
|
||||
| `--model <model>` | string | no | Model ID (default: fun-asr) |
|
||||
| `--language <lang>` | string | no | Language hint (e.g. zh, en, ja) |
|
||||
| `--diarization` | switch | no | Enable automatic speaker diarization |
|
||||
| `--speaker-count <n>` | number | no | Expected number of speakers (requires --diarization) |
|
||||
| `--vocabulary-id <id>` | string | no | Hot-word vocabulary ID for improved accuracy |
|
||||
| `--channel-id <n>` | number | no | Audio channel ID (default: 0) |
|
||||
| `--out <path>` | string | no | Save full transcription result to JSON file |
|
||||
| `--async` | switch | no | Return async task id without waiting |
|
||||
| `--poll-interval <seconds>` | number | no | Polling interval in seconds (default: 2) |
|
||||
| `--api-key <key>` | string | no | API key |
|
||||
| `--base-url <url>` | string | no | API base URL |
|
||||
| Flag | Type | Required | Description |
|
||||
| --------------------------- | ------ | -------- | ------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `--url <url>` | array | yes | Audio file URL or local file path (repeatable, max 100) |
|
||||
| `--model <model>` | string | no | Model ID (default: fun-asr). Async: fun-asr / _-filetrans / paraformer-_; sync: qwen3-asr-flash* / fun-asr-flash* / qwen-audio-\*-asr-flash |
|
||||
| `--language <lang>` | string | no | Language hint (e.g. zh, en, ja). Async & input-audio sync: language_hints; qwen3 sync: asr_options.language |
|
||||
| `--diarization` | switch | no | Enable automatic speaker diarization |
|
||||
| `--speaker-count <n>` | number | no | Expected number of speakers (requires --diarization) |
|
||||
| `--vocabulary-id <id>` | string | no | Hot-word vocabulary ID for improved accuracy |
|
||||
| `--channel-id <n>` | number | no | Audio channel ID (default: 0) |
|
||||
| `--out <path>` | string | no | Save full transcription result to JSON file |
|
||||
| `--async` | switch | no | Return async task id without waiting |
|
||||
| `--poll-interval <seconds>` | number | no | Polling interval in seconds (default: 2) |
|
||||
| `--api-key <key>` | string | no | API key |
|
||||
| `--base-url <url>` | string | no | API base URL |
|
||||
|
||||
#### Examples
|
||||
|
||||
@@ -69,6 +69,10 @@ bl speech recognize --url https://example.com/audio.mp3 --out result.json
|
||||
bl speech recognize --url https://example.com/audio.mp3 --async --quiet
|
||||
```
|
||||
|
||||
```bash
|
||||
bl speech recognize --url https://example.com/audio.mp3 --model qwen-audio-3.0-asr-flash --language en
|
||||
```
|
||||
|
||||
### `bl speech synthesize`
|
||||
|
||||
| Field | Value |
|
||||
|
||||
Reference in New Issue
Block a user