mirror of
https://github.com/modelstudioai/cli.git
synced 2026-09-14 19:49:23 +08:00
fix(speech): handle qwen3-filetrans singular result.transcription_url
Normalize async ASR transcription items so waiting mode downloads text and --out works without changing shared media task types.
This commit is contained in:
@@ -15,6 +15,7 @@ import {
|
||||
resolveAsrApi,
|
||||
buildAsrFlashRequest,
|
||||
buildAsyncAsrLanguageFields,
|
||||
collectAsrTranscriptionItems,
|
||||
extractAsrFlashText,
|
||||
type AsrApiRoute,
|
||||
type AsrFlashFamily,
|
||||
@@ -79,7 +80,7 @@ function assertSyncFlashFlagsAllowed(
|
||||
const unsupported: string[] = [];
|
||||
if (flags.diarization === true) unsupported.push("--diarization");
|
||||
if (flags.speakerCount !== undefined) unsupported.push("--speaker-count");
|
||||
// qwen3 sync Flash 不走 vocabulary_id;input-audio Flash(fun-asr-flash* / qwen-audio-*-asr-flash)官方支持
|
||||
// qwen3 sync Flash does not use vocabulary_id; input-audio Flash (fun-asr-flash* / qwen-audio-*-asr-flash) does
|
||||
if (flashFamily === "qwen3" && flags.vocabularyId !== undefined) {
|
||||
unsupported.push("--vocabulary-id");
|
||||
}
|
||||
@@ -308,7 +309,7 @@ async function handleAsyncMode(
|
||||
},
|
||||
});
|
||||
|
||||
const results = result.output.results ?? [];
|
||||
const results = collectAsrTranscriptionItems(result.output);
|
||||
|
||||
if (results.length === 0) {
|
||||
emitResult({ task_id: taskId, status: result.output.task_status }, format);
|
||||
|
||||
@@ -97,7 +97,7 @@ describe("e2e: speech recognize", () => {
|
||||
});
|
||||
|
||||
test("speech recognize realtime 模型报用法错误", async () => {
|
||||
// 使用 --dry-run:跳过 auth,避免 CI 无密钥时先以 AUTH(3) 退出
|
||||
// Use --dry-run to skip auth so CI without API keys still hits USAGE(2)
|
||||
const { stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [
|
||||
"speech",
|
||||
"recognize",
|
||||
@@ -188,6 +188,87 @@ describe("e2e: speech recognize", () => {
|
||||
expect(exitCode).toBe(2);
|
||||
expect(stderr).toMatch(/exactly one --url|sync Flash/i);
|
||||
});
|
||||
|
||||
test("speech recognize qwen3-filetrans 轮询成功后下载 result.transcription_url", async () => {
|
||||
const server = http.createServer((request, response) => {
|
||||
const url = request.url ?? "";
|
||||
const chunks: Buffer[] = [];
|
||||
request.on("data", (chunk: Buffer) => chunks.push(chunk));
|
||||
request.on("end", () => {
|
||||
response.writeHead(200, { "Content-Type": "application/json" });
|
||||
if (url.startsWith("/api/v1/services/audio/asr/transcription")) {
|
||||
response.end(
|
||||
JSON.stringify({
|
||||
output: { task_id: "task-qwen3", task_status: "PENDING" },
|
||||
request_id: "req-submit",
|
||||
}),
|
||||
);
|
||||
return;
|
||||
}
|
||||
if (url.startsWith("/api/v1/tasks/")) {
|
||||
const address = server.address() as AddressInfo;
|
||||
response.end(
|
||||
JSON.stringify({
|
||||
output: {
|
||||
task_id: "task-qwen3",
|
||||
task_status: "SUCCEEDED",
|
||||
result: {
|
||||
transcription_url: `http://127.0.0.1:${address.port}/transcription.json`,
|
||||
},
|
||||
},
|
||||
request_id: "req-poll",
|
||||
}),
|
||||
);
|
||||
return;
|
||||
}
|
||||
if (url.startsWith("/transcription.json")) {
|
||||
response.end(
|
||||
JSON.stringify({
|
||||
file_url: "https://example.com/a.wav",
|
||||
transcripts: [{ text: "你好世界", sentences: [{ text: "你好世界" }] }],
|
||||
}),
|
||||
);
|
||||
return;
|
||||
}
|
||||
response.writeHead(404);
|
||||
response.end(JSON.stringify({ message: `unexpected path: ${url}` }));
|
||||
});
|
||||
});
|
||||
await new Promise<void>((resolve) => server.listen(0, "127.0.0.1", resolve));
|
||||
const address = server.address() as AddressInfo;
|
||||
const outDir = makeE2eOutputDir("speech-recognize-qwen3-filetrans");
|
||||
const outPath = join(outDir, "result.json");
|
||||
|
||||
try {
|
||||
const { stdout, stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [
|
||||
"speech",
|
||||
"recognize",
|
||||
"--model",
|
||||
"qwen3-asr-flash-filetrans",
|
||||
"--url",
|
||||
"https://example.com/a.wav",
|
||||
"--language",
|
||||
"zh",
|
||||
"--api-key",
|
||||
"sk-e2e-placeholder",
|
||||
"--base-url",
|
||||
`http://127.0.0.1:${address.port}`,
|
||||
"--poll-interval",
|
||||
"1",
|
||||
"--out",
|
||||
outPath,
|
||||
"--quiet",
|
||||
]);
|
||||
|
||||
expect(exitCode, stderr).toBe(0);
|
||||
expect(stdout).toContain("你好世界");
|
||||
expect(JSON.parse(readFileSync(outPath, "utf8"))).toMatchObject({
|
||||
transcripts: [{ text: "你好世界" }],
|
||||
});
|
||||
} finally {
|
||||
await new Promise<void>((resolve) => server.close(() => resolve()));
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe.skipIf(!isBailianE2EMediaEnabled() || !isDashScopeE2EReady())(
|
||||
|
||||
@@ -136,7 +136,7 @@ export function resolveAsrApi(model: string): AsrApiRoute {
|
||||
|
||||
/** Infer audio container hint for input-audio Flash `parameters.format`. */
|
||||
export function inferAudioFormatHint(audioUrl: string): string {
|
||||
// data URI:data:audio/mpeg;base64,... → mp3;data:audio/x-wav;... → wav
|
||||
// data URI: data:audio/mpeg;base64,... → mp3; data:audio/x-wav;... → wav
|
||||
const dataType = /^data:audio\/([^;,]+)/i.exec(audioUrl)?.[1]?.toLowerCase();
|
||||
if (dataType) {
|
||||
if (dataType === "mpeg") return "mp3";
|
||||
@@ -156,14 +156,14 @@ export interface BuildAsrFlashRequestOpts {
|
||||
model: string;
|
||||
audioUrl: string;
|
||||
language?: string;
|
||||
/** 预编译热词 ID;仅 input-audio Flash(fun-asr-flash* / qwen-audio-*-asr-flash)官方支持 */
|
||||
/** Precompiled hotword vocabulary ID; supported for input-audio Flash (fun-asr-flash* / qwen-audio-*-asr-flash). */
|
||||
vocabularyId?: string;
|
||||
flashFamily: AsrFlashFamily;
|
||||
}
|
||||
|
||||
/**
|
||||
* 按异步路由的 language 字段风格构造语种参数。
|
||||
* qwen3-asr-flash-filetrans* → `language`;其余异步模型 → `language_hints`。
|
||||
* Build language fields for async ASR routes.
|
||||
* qwen3-asr-flash-filetrans* → `language`; other async models → `language_hints`.
|
||||
*/
|
||||
export function buildAsyncAsrLanguageFields(
|
||||
languageStyle: "language_hints" | "language",
|
||||
@@ -181,7 +181,7 @@ export function buildAsrFlashRequest(opts: BuildAsrFlashRequestOpts): Record<str
|
||||
const { model, audioUrl, language, vocabularyId, flashFamily } = opts;
|
||||
|
||||
if (flashFamily === "input-audio") {
|
||||
// 与官方 Qwen-Audio / Fun-ASR-Flash 文档一致:语种走 language_hints,热词走 vocabulary_id
|
||||
// Match official Qwen-Audio / Fun-ASR-Flash docs: language_hints + vocabulary_id
|
||||
const parameters: Record<string, unknown> = {
|
||||
format: inferAudioFormatHint(audioUrl),
|
||||
sample_rate: "16000",
|
||||
@@ -294,3 +294,34 @@ export function extractAsrFlashText(
|
||||
}
|
||||
return texts.join("");
|
||||
}
|
||||
|
||||
/**
|
||||
* Normalize async ASR task transcription items:
|
||||
* - classic models: `output.results[]`
|
||||
* - qwen3-asr-flash-filetrans*: `output.result.transcription_url`
|
||||
*/
|
||||
export function collectAsrTranscriptionItems(output: {
|
||||
results?: Array<{
|
||||
file_url?: string;
|
||||
transcription_url?: string;
|
||||
subtask_status?: string;
|
||||
code?: string;
|
||||
message?: string;
|
||||
}>;
|
||||
result?: { transcription_url?: string };
|
||||
}): Array<{
|
||||
file_url?: string;
|
||||
transcription_url?: string;
|
||||
subtask_status?: string;
|
||||
code?: string;
|
||||
message?: string;
|
||||
}> {
|
||||
if (output.results && output.results.length > 0) {
|
||||
return output.results;
|
||||
}
|
||||
const transcriptionUrl = output.result?.transcription_url;
|
||||
if (typeof transcriptionUrl === "string" && transcriptionUrl.length > 0) {
|
||||
return [{ transcription_url: transcriptionUrl, subtask_status: "SUCCEEDED" }];
|
||||
}
|
||||
return [];
|
||||
}
|
||||
|
||||
@@ -37,6 +37,7 @@ export {
|
||||
export {
|
||||
buildAsrFlashRequest,
|
||||
buildAsyncAsrLanguageFields,
|
||||
collectAsrTranscriptionItems,
|
||||
extractAsrFlashText,
|
||||
inferAudioFormatHint,
|
||||
resolveAsrApi,
|
||||
|
||||
@@ -538,9 +538,9 @@ export interface DashScopeASRRequest {
|
||||
};
|
||||
parameters?: {
|
||||
channel_id?: number[];
|
||||
/** fun-asr / paraformer / qwen-audio filetrans 等经典异步模型 */
|
||||
/** Classic async models (fun-asr / paraformer / qwen-audio filetrans, etc.) */
|
||||
language_hints?: string[];
|
||||
/** qwen3-asr-flash-filetrans* 使用单数字段 language */
|
||||
/** qwen3-asr-flash-filetrans* uses singular `language` */
|
||||
language?: string;
|
||||
diarization_enabled?: boolean;
|
||||
speaker_count?: number;
|
||||
@@ -548,22 +548,31 @@ export interface DashScopeASRRequest {
|
||||
};
|
||||
}
|
||||
|
||||
export interface DashScopeASRTranscriptionItem {
|
||||
file_url?: string;
|
||||
transcription_url?: string;
|
||||
subtask_status?: string;
|
||||
code?: string;
|
||||
message?: string;
|
||||
}
|
||||
|
||||
export interface DashScopeASRTaskResult {
|
||||
output: {
|
||||
task_id: string;
|
||||
task_status: "PENDING" | "RUNNING" | "SUCCEEDED" | "FAILED" | "UNKNOWN";
|
||||
results?: Array<{
|
||||
file_url?: string;
|
||||
/** Multi-file async results (fun-asr / paraformer / qwen-audio filetrans, etc.) */
|
||||
results?: DashScopeASRTranscriptionItem[];
|
||||
/** Singular result returned by qwen3-asr-flash-filetrans* on success */
|
||||
result?: {
|
||||
transcription_url?: string;
|
||||
subtask_status?: string;
|
||||
code?: string;
|
||||
message?: string;
|
||||
}>;
|
||||
};
|
||||
task_metrics?: {
|
||||
TOTAL: number;
|
||||
SUCCEEDED: number;
|
||||
FAILED: number;
|
||||
};
|
||||
code?: string;
|
||||
message?: string;
|
||||
};
|
||||
usage?: Record<string, unknown>;
|
||||
request_id: string;
|
||||
|
||||
@@ -48,6 +48,7 @@ export type {
|
||||
ChatTool,
|
||||
DashScopeASRRequest,
|
||||
DashScopeASRTaskResult,
|
||||
DashScopeASRTranscriptionItem,
|
||||
DashScopeAsyncResponse,
|
||||
DashScopeImageRequest,
|
||||
DashScopeImageSyncResponse,
|
||||
|
||||
@@ -2,6 +2,7 @@ import { expect, test } from "vite-plus/test";
|
||||
import {
|
||||
buildAsrFlashRequest,
|
||||
buildAsyncAsrLanguageFields,
|
||||
collectAsrTranscriptionItems,
|
||||
extractAsrFlashText,
|
||||
inferAudioFormatHint,
|
||||
resolveAsrApi,
|
||||
@@ -194,3 +195,19 @@ test("extractAsrFlashText reads qwen3 choices and input-audio text fields", () =
|
||||
),
|
||||
).toBe("nested sentence");
|
||||
});
|
||||
|
||||
test("collectAsrTranscriptionItems prefers results[] then singular result", () => {
|
||||
expect(
|
||||
collectAsrTranscriptionItems({
|
||||
results: [{ transcription_url: "https://example.com/a.json", file_url: "https://a.wav" }],
|
||||
}),
|
||||
).toEqual([{ transcription_url: "https://example.com/a.json", file_url: "https://a.wav" }]);
|
||||
|
||||
expect(
|
||||
collectAsrTranscriptionItems({
|
||||
result: { transcription_url: "https://example.com/qwen3.json" },
|
||||
}),
|
||||
).toEqual([{ transcription_url: "https://example.com/qwen3.json", subtask_status: "SUCCEEDED" }]);
|
||||
|
||||
expect(collectAsrTranscriptionItems({})).toEqual([]);
|
||||
});
|
||||
|
||||
@@ -13,6 +13,7 @@ import {
|
||||
resolveAsrApi,
|
||||
buildAsrFlashRequest,
|
||||
buildAsyncAsrLanguageFields,
|
||||
collectAsrTranscriptionItems,
|
||||
extractAsrFlashText,
|
||||
stripUndefined,
|
||||
resolveBooleanFlag,
|
||||
@@ -27,6 +28,7 @@ import {
|
||||
type DashScopeTTSRequest,
|
||||
type DashScopeTTSResponse,
|
||||
type DashScopeASRRequest,
|
||||
type DashScopeASRTaskResult,
|
||||
type ChatMessageContent,
|
||||
isLocalFile,
|
||||
} from "bailian-cli-core";
|
||||
@@ -600,7 +602,7 @@ export async function speechRecognize(
|
||||
const unsupportedFlags: string[] = [];
|
||||
if (input.diarization) unsupportedFlags.push("diarization");
|
||||
if (input["speaker-count"] !== undefined) unsupportedFlags.push("speaker-count");
|
||||
// input-audio Flash 官方支持 vocabulary_id;qwen3 sync Flash 不支持
|
||||
// input-audio Flash supports vocabulary_id; qwen3 sync Flash does not
|
||||
if (route.flashFamily === "qwen3" && input["vocabulary-id"] !== undefined) {
|
||||
unsupportedFlags.push("vocabulary-id");
|
||||
}
|
||||
@@ -693,7 +695,65 @@ export async function speechRecognize(
|
||||
const pollIntervalMs = (input["poll-interval"] ?? 2) * 1000;
|
||||
const timeoutMs = (ctx.timeoutSeconds ?? 300) * 1000;
|
||||
|
||||
return await pollTaskWithOptions(env, taskId, pollIntervalMs, timeoutMs, ctx);
|
||||
// ASR polling reads original output, avoids generic flatten (avoids transcription_url polluting media urls)
|
||||
const asrTask = await pollAsrTaskWithOptions(env, taskId, pollIntervalMs, timeoutMs, ctx);
|
||||
const transcriptionItems = collectAsrTranscriptionItems(asrTask.output);
|
||||
|
||||
const base: Record<string, unknown> = {
|
||||
task_id: asrTask.output.task_id,
|
||||
task_status: asrTask.output.task_status,
|
||||
request_id: asrTask.request_id,
|
||||
mode: "async",
|
||||
model,
|
||||
};
|
||||
if (asrTask.output.results) base.results = asrTask.output.results;
|
||||
if (asrTask.output.result) {
|
||||
base.result = asrTask.output.result;
|
||||
if (typeof asrTask.output.result.transcription_url === "string") {
|
||||
base.transcription_url = asrTask.output.result.transcription_url;
|
||||
}
|
||||
}
|
||||
if (asrTask.output.task_metrics) base.task_metrics = asrTask.output.task_metrics;
|
||||
if (asrTask.usage) base.usage = asrTask.usage;
|
||||
|
||||
if (transcriptionItems.length === 0) {
|
||||
return base;
|
||||
}
|
||||
|
||||
const texts: string[] = [];
|
||||
const transcripts: Record<string, unknown>[] = [];
|
||||
for (const item of transcriptionItems) {
|
||||
if (!item.transcription_url) continue;
|
||||
const transRes = await fetch(item.transcription_url, { signal: ctx.signal });
|
||||
if (!transRes.ok) {
|
||||
throw new PipelineError(
|
||||
"async_task_failed",
|
||||
`Failed to download transcription: HTTP ${transRes.status}`,
|
||||
{ step: "speech/recognize", details: { taskId, url: item.transcription_url } },
|
||||
);
|
||||
}
|
||||
const transData = (await transRes.json()) as Record<string, unknown>;
|
||||
transcripts.push(transData);
|
||||
const transcriptList = transData.transcripts as
|
||||
| Array<{ text?: string; sentences?: Array<{ text?: string }> }>
|
||||
| undefined;
|
||||
if (!transcriptList?.length) continue;
|
||||
for (const transcript of transcriptList) {
|
||||
if (transcript.sentences?.length) {
|
||||
for (const sentence of transcript.sentences) {
|
||||
if (sentence.text) texts.push(sentence.text);
|
||||
}
|
||||
} else if (transcript.text) {
|
||||
texts.push(transcript.text);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
...base,
|
||||
text: texts.join("\n"),
|
||||
transcripts,
|
||||
};
|
||||
}
|
||||
|
||||
// --- Shared: task polling ---
|
||||
@@ -719,7 +779,7 @@ function flattenTaskResponse(resp: DashScopeTaskResponse): Record<string, unknow
|
||||
if (urls.length > 0) flat.urls = urls;
|
||||
}
|
||||
if (output.results) {
|
||||
const urls = output.results.map((r) => r.url).filter(Boolean);
|
||||
const urls = output.results.map((item) => item.url).filter(Boolean);
|
||||
if (urls.length > 0 && !flat.urls) flat.urls = urls;
|
||||
}
|
||||
if (output.task_metrics) flat.task_metrics = output.task_metrics;
|
||||
@@ -737,13 +797,13 @@ async function pollTask(
|
||||
return await pollTaskWithOptions(env, taskId, pollIntervalMs, timeoutMs, ctx);
|
||||
}
|
||||
|
||||
async function pollTaskWithOptions(
|
||||
async function pollUntilSucceeded(
|
||||
env: PipelineEnv,
|
||||
taskId: string,
|
||||
pollIntervalMs: number,
|
||||
timeoutMs: number,
|
||||
ctx?: StepContext,
|
||||
): Promise<Record<string, unknown>> {
|
||||
): Promise<DashScopeTaskResponse> {
|
||||
const started = Date.now();
|
||||
let attempt = 0;
|
||||
|
||||
@@ -766,7 +826,7 @@ async function pollTaskWithOptions(
|
||||
const status = result.output.task_status;
|
||||
|
||||
if (status === "SUCCEEDED") {
|
||||
return flattenTaskResponse(result);
|
||||
return result;
|
||||
}
|
||||
|
||||
if (status === "FAILED") {
|
||||
@@ -797,6 +857,33 @@ async function pollTaskWithOptions(
|
||||
}
|
||||
}
|
||||
|
||||
async function pollTaskWithOptions(
|
||||
env: PipelineEnv,
|
||||
taskId: string,
|
||||
pollIntervalMs: number,
|
||||
timeoutMs: number,
|
||||
ctx?: StepContext,
|
||||
): Promise<Record<string, unknown>> {
|
||||
return flattenTaskResponse(await pollUntilSucceeded(env, taskId, pollIntervalMs, timeoutMs, ctx));
|
||||
}
|
||||
|
||||
/** ASR task polling: preserve original output (includes results[] / result.transcription_url). */
|
||||
async function pollAsrTaskWithOptions(
|
||||
env: PipelineEnv,
|
||||
taskId: string,
|
||||
pollIntervalMs: number,
|
||||
timeoutMs: number,
|
||||
ctx?: StepContext,
|
||||
): Promise<DashScopeASRTaskResult> {
|
||||
return (await pollUntilSucceeded(
|
||||
env,
|
||||
taskId,
|
||||
pollIntervalMs,
|
||||
timeoutMs,
|
||||
ctx,
|
||||
)) as DashScopeASRTaskResult;
|
||||
}
|
||||
|
||||
function delay(ms: number, signal?: AbortSignal): Promise<void> {
|
||||
if (!signal) return new Promise((resolve) => setTimeout(resolve, ms));
|
||||
return new Promise((resolve, reject) => {
|
||||
|
||||
@@ -128,3 +128,57 @@ test("pipeline speechRecognize rejects multiple urls for sync flash", async () =
|
||||
).rejects.toBeInstanceOf(PipelineError);
|
||||
expect(captured).toHaveLength(0);
|
||||
});
|
||||
|
||||
test("pipeline speechRecognize downloads qwen3 singular result.transcription_url", async () => {
|
||||
const originalFetch = globalThis.fetch;
|
||||
const transcriptionUrl = "https://example.com/transcription.json";
|
||||
globalThis.fetch = (async (input: RequestInfo | URL) => {
|
||||
const url = typeof input === "string" ? input : input instanceof URL ? input.href : input.url;
|
||||
expect(url).toBe(transcriptionUrl);
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
transcripts: [{ text: "pipeline hello", sentences: [{ text: "pipeline hello" }] }],
|
||||
}),
|
||||
{ status: 200, headers: { "Content-Type": "application/json" } },
|
||||
);
|
||||
}) as typeof fetch;
|
||||
|
||||
try {
|
||||
const { env, captured } = makeEnv(async (opts) => {
|
||||
if (opts.async || opts.method === "POST") {
|
||||
return { output: { task_id: "task-1", task_status: "PENDING" } };
|
||||
}
|
||||
return {
|
||||
output: {
|
||||
task_id: "task-1",
|
||||
task_status: "SUCCEEDED",
|
||||
result: { transcription_url: transcriptionUrl },
|
||||
},
|
||||
request_id: "r1",
|
||||
};
|
||||
});
|
||||
|
||||
const result = (await speechRecognize(
|
||||
env,
|
||||
{
|
||||
url: "https://example.com/a.wav",
|
||||
model: "qwen3-asr-flash-filetrans",
|
||||
"poll-interval": 0,
|
||||
},
|
||||
makeCtx(),
|
||||
)) as {
|
||||
mode?: string;
|
||||
text?: string;
|
||||
transcription_url?: string;
|
||||
result?: { transcription_url?: string };
|
||||
};
|
||||
|
||||
expect(captured[0]?.async).toBe(true);
|
||||
expect(result.mode).toBe("async");
|
||||
expect(result.text).toBe("pipeline hello");
|
||||
expect(result.transcription_url).toBe(transcriptionUrl);
|
||||
expect(result.result?.transcription_url).toBe(transcriptionUrl);
|
||||
} finally {
|
||||
globalThis.fetch = originalFetch;
|
||||
}
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user