test(speech): harden flash ASR contract coverage and docs

Add SSE disable header, data-URI format inference, broader response text
parsing, HTTP contract e2e, pipeline routing tests, and ASR model selection
guidance in bailian-gen.
This commit is contained in:
clh02467605
2026-08-13 14:25:47 +08:00
parent 9379da7a4c
commit e244771ee9
7 changed files with 256 additions and 5 deletions
@@ -248,6 +248,7 @@ async function handleSyncFlashMode(
const response = await client.requestJson<Record<string, unknown>>({
path: route.path,
method: "POST",
headers: { "X-DashScope-SSE": "disable" },
body,
});
@@ -1,4 +1,6 @@
import { readFileSync } from "node:fs";
import http from "node:http";
import type { AddressInfo } from "node:net";
import { join } from "node:path";
import { describe, expect, test } from "vite-plus/test";
import {
@@ -109,6 +111,83 @@ describe("e2e: speech recognize", () => {
expect(exitCode).toBe(2);
expect(stderr).toMatch(/realtime|WebSocket|unsupported/i);
});
test("speech recognize flash 真实请求走 sync endpoint 并落盘 --out", async () => {
let requestPath = "";
let requestBody: Record<string, unknown> = {};
let sseHeader: string | undefined;
const server = http.createServer((request, response) => {
const chunks: Buffer[] = [];
request.on("data", (chunk: Buffer) => chunks.push(chunk));
request.on("end", () => {
requestPath = request.url ?? "";
requestBody = JSON.parse(Buffer.concat(chunks).toString("utf8")) as Record<string, unknown>;
sseHeader = request.headers["x-dashscope-sse"] as string | undefined;
response.writeHead(200, { "Content-Type": "application/json" });
response.end(
JSON.stringify({
output: { text: "flash recognition works" },
request_id: "request-146",
}),
);
});
});
await new Promise<void>((resolve) => server.listen(0, "127.0.0.1", resolve));
const address = server.address() as AddressInfo;
const outDir = makeE2eOutputDir("speech-recognize-flash-sync");
const outPath = join(outDir, "result.json");
try {
const { stdout, stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [
"speech",
"recognize",
"--model",
"fun-asr-flash-2026-06-15",
"--url",
"https://example.com/sample.wav",
"--api-key",
"sk-e2e-placeholder",
"--base-url",
`http://127.0.0.1:${address.port}`,
"--out",
outPath,
"--quiet",
]);
expect(exitCode, stderr).toBe(0);
expect(stdout).toContain("flash recognition works");
expect(requestPath).toBe("/api/v1/services/aigc/multimodal-generation/generation");
expect(sseHeader).toBe("disable");
expect(requestBody).toMatchObject({
model: "fun-asr-flash-2026-06-15",
parameters: { format: "wav" },
});
expect(JSON.parse(readFileSync(outPath, "utf8"))).toMatchObject({
output: { text: "flash recognition works" },
request_id: "request-146",
});
} finally {
await new Promise<void>((resolve) => server.close(() => resolve()));
}
});
test("speech recognize flash 多 --url 在发请求前报用法错误", async () => {
const { stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [
"speech",
"recognize",
"--model",
"qwen-audio-3.0-asr-flash",
"--url",
"https://example.com/a.wav",
"--url",
"https://example.com/b.wav",
"--dry-run",
"--quiet",
]);
expect(exitCode).toBe(2);
expect(stderr).toMatch(/exactly one --url|sync Flash/i);
});
});
describe.skipIf(!isBailianE2EMediaEnabled() || !isDashScopeE2EReady())(
+18 -5
View File
@@ -136,7 +136,15 @@ export function resolveAsrApi(model: string): AsrApiRoute {
/** Infer audio container hint for input-audio Flash `parameters.format`. */
export function inferAudioFormatHint(audioUrl: string): string {
const pathPart = audioUrl.split("?")[0] ?? audioUrl;
// data URI:data:audio/mpeg;base64,... → mp3;data:audio/x-wav;... → wav
const dataType = /^data:audio\/([^;,]+)/i.exec(audioUrl)?.[1]?.toLowerCase();
if (dataType) {
if (dataType === "mpeg") return "mp3";
if (dataType === "x-wav" || dataType === "wave") return "wav";
return dataType;
}
const pathPart = audioUrl.split(/[?#]/, 1)[0] ?? audioUrl;
const match = pathPart.match(/\.([a-zA-Z0-9]+)$/);
const extension = match?.[1]?.toLowerCase();
if (!extension) return "wav";
@@ -232,7 +240,8 @@ export function buildAsrFlashRequest(opts: BuildAsrFlashRequestOpts): Record<str
/**
* Extract recognition text from a sync Flash ASR response.
* Qwen3 uses choices[].message.content; input-audio Flash uses output.text.
* Qwen3 uses choices[].message.content; input-audio Flash uses output.text /
* output.sentence.text / output.output.sentence.text.
*/
export function extractAsrFlashText(
response: Record<string, unknown>,
@@ -245,10 +254,14 @@ export function extractAsrFlashText(
if (typeof output.text === "string" && output.text.length > 0) {
return output.text;
}
const topSentence = output.sentence as Record<string, unknown> | undefined;
if (typeof topSentence?.text === "string" && topSentence.text.length > 0) {
return topSentence.text;
}
const nested = output.output as Record<string, unknown> | undefined;
const sentence = nested?.sentence as Record<string, unknown> | undefined;
if (typeof sentence?.text === "string") {
return sentence.text;
const nestedSentence = nested?.sentence as Record<string, unknown> | undefined;
if (typeof nestedSentence?.text === "string") {
return nestedSentence.text;
}
return "";
}
+25
View File
@@ -92,6 +92,9 @@ test("inferAudioFormatHint reads extension from url", () => {
expect(inferAudioFormatHint("oss://bucket/path/file.WAV")).toBe("wav");
expect(inferAudioFormatHint("https://example.com/a.mpeg?x=1")).toBe("mp3");
expect(inferAudioFormatHint("https://example.com/noext")).toBe("wav");
expect(inferAudioFormatHint("data:audio/mpeg;base64,AAA")).toBe("mp3");
expect(inferAudioFormatHint("data:audio/x-wav;base64,AAA")).toBe("wav");
expect(inferAudioFormatHint("data:audio/ogg;codecs=opus;base64,AAA")).toBe("ogg");
});
test("buildAsrFlashRequest shapes qwen3 and input-audio bodies", () => {
@@ -168,4 +171,26 @@ test("extractAsrFlashText reads qwen3 choices and input-audio text fields", () =
"input-audio",
),
).toBe("Hello World");
expect(
extractAsrFlashText(
{
output: {
sentence: { text: "top-level sentence" },
},
},
"input-audio",
),
).toBe("top-level sentence");
expect(
extractAsrFlashText(
{
output: {
output: { sentence: { text: "nested sentence" } },
},
},
"input-audio",
),
).toBe("nested sentence");
});
@@ -651,6 +651,7 @@ export async function speechRecognize(
const response = await env.client.requestJson<Record<string, unknown>>({
path: route.path,
method: "POST",
headers: { "X-DashScope-SSE": "disable" },
body,
signal: ctx.signal,
});
@@ -0,0 +1,130 @@
import { expect, test } from "vite-plus/test";
import type { Client } from "bailian-cli-core";
import { PipelineError } from "../src/pipeline/errors.ts";
import type { PipelineEnv } from "../src/pipeline/bl-config.ts";
import { speechRecognize } from "../src/pipeline/steps/bl-api.ts";
import type { StepContext } from "../src/pipeline/types.ts";
type CapturedRequest = {
path?: string;
method?: string;
headers?: Record<string, string>;
body?: Record<string, unknown>;
async?: boolean;
};
function makeEnv(requestJsonImpl?: (opts: CapturedRequest) => Promise<unknown>): {
env: PipelineEnv;
captured: CapturedRequest[];
} {
const captured: CapturedRequest[] = [];
const client = {
uploadFile: async (source: string) => source,
requestJson: async (opts: CapturedRequest) => {
captured.push(opts);
if (requestJsonImpl) return requestJsonImpl(opts);
return { output: { text: "ok" } };
},
} as unknown as Client;
return {
env: {
client,
settings: { quiet: true, output: "json" } as PipelineEnv["settings"],
},
captured,
};
}
function makeCtx(): StepContext {
return { dryRun: false, signal: new AbortController().signal };
}
test("pipeline speechRecognize routes input-audio flash to sync multimodal endpoint", async () => {
const { env, captured } = makeEnv();
const result = (await speechRecognize(
env,
{
url: "https://example.com/a.wav",
model: "qwen-audio-3.0-asr-flash",
language: "en",
"vocabulary-id": "vocab-1",
},
makeCtx(),
)) as { mode?: string; text?: string };
expect(result.mode).toBe("sync");
expect(result.text).toBe("ok");
expect(captured).toHaveLength(1);
expect(captured[0]?.path).toBe("/api/v1/services/aigc/multimodal-generation/generation");
expect(captured[0]?.headers?.["X-DashScope-SSE"]).toBe("disable");
expect(captured[0]?.body).toMatchObject({
model: "qwen-audio-3.0-asr-flash",
parameters: {
format: "wav",
language_hints: ["en"],
vocabulary_id: "vocab-1",
},
});
});
test("pipeline speechRecognize maps qwen3-filetrans language to parameters.language", async () => {
const { env, captured } = makeEnv(async (opts) => {
if (opts.async || opts.method === "POST") {
return { output: { task_id: "task-1", task_status: "PENDING" } };
}
return {
output: { task_id: "task-1", task_status: "SUCCEEDED", results: [] },
request_id: "r1",
};
});
await speechRecognize(
env,
{
url: "https://example.com/a.wav",
model: "qwen3-asr-flash-filetrans",
language: "zh",
"poll-interval": 0,
},
makeCtx(),
);
expect(captured[0]?.path).toBe("/api/v1/services/audio/asr/transcription");
expect(captured[0]?.async).toBe(true);
expect(captured[0]?.body).toMatchObject({
model: "qwen3-asr-flash-filetrans",
input: { file_url: "https://example.com/a.wav" },
parameters: { language: "zh" },
});
expect(
(captured[0]?.body?.parameters as Record<string, unknown> | undefined)?.language_hints,
).toBeUndefined();
});
test("pipeline speechRecognize rejects realtime models before requesting", async () => {
const { env, captured } = makeEnv();
await expect(
speechRecognize(
env,
{ url: "https://example.com/a.wav", model: "qwen3-asr-flash-realtime" },
makeCtx(),
),
).rejects.toBeInstanceOf(PipelineError);
expect(captured).toHaveLength(0);
});
test("pipeline speechRecognize rejects multiple urls for sync flash", async () => {
const { env, captured } = makeEnv();
await expect(
speechRecognize(
env,
{
url: ["https://example.com/a.wav", "https://example.com/b.wav"],
model: "fun-asr-flash-2026-06-15",
},
makeCtx(),
),
).rejects.toBeInstanceOf(PipelineError);
expect(captured).toHaveLength(0);
});
+2
View File
@@ -39,6 +39,8 @@ description: >-
| A/V understanding (files the host can't play) | `bl omni --video` / `--audio` | `qwen3.5-omni-plus` |
| Image/video describe (user names Bailian) | `bl vision describe` | `qwen-vl-max`; host-first for plain image Q&A |
For ASR model selection, keep `fun-asr` (or other `*-filetrans`) for long recordings, repeated files, speaker diarization, or asynchronous task IDs. For one local or remote audio file up to about five minutes when the user asks for low-latency Flash models, use `--model fun-asr-flash-2026-06-15`, `--model qwen-audio-3.0-asr-flash`, or `--model qwen3-asr-flash`. Flash recognition is synchronous and accepts exactly one file per call.
Flags, usage, and examples: see [`reference/`](reference/index.md) or `bl <command> --help` — do not guess flags.
## Local files (mandatory)