diff --git a/packages/cli/src/commands.ts b/packages/cli/src/commands.ts index 652da45..cf0ebb9 100644 --- a/packages/cli/src/commands.ts +++ b/packages/cli/src/commands.ts @@ -45,6 +45,7 @@ import { usageFreetier, usageStats, usageSummary, + usageTokenPlan, pipelineRun, pipelineValidate, advisorRecommend, @@ -164,6 +165,7 @@ export const commands: Record = { "usage freetier": usageFreetier, "usage stats": usageStats, "usage summary": usageSummary, + "usage token-plan": usageTokenPlan, "pipeline run": pipelineRun, "pipeline validate": pipelineValidate, "advisor recommend": advisorRecommend, diff --git a/packages/commands/src/commands/speech/recognize.ts b/packages/commands/src/commands/speech/recognize.ts index 5af606b..0b047e7 100644 --- a/packages/commands/src/commands/speech/recognize.ts +++ b/packages/commands/src/commands/speech/recognize.ts @@ -12,6 +12,13 @@ import { stripUndefined, taskPath, speechRecognizePath, + resolveAsrApi, + buildAsrFlashRequest, + buildAsyncAsrLanguageFields, + collectAsrTranscriptionItems, + extractAsrFlashText, + type AsrApiRoute, + type AsrFlashFamily, type OutputFormat, type FlagsDef, type ParsedFlags, @@ -27,8 +34,18 @@ const RECOGNIZE_FLAGS = { description: "Audio file URL or local file path (repeatable, max 100)", required: true, }, - model: { type: "string", valueHint: "", description: "Model ID (default: fun-asr)" }, - language: { type: "string", valueHint: "", description: "Language hint (e.g. zh, en, ja)" }, + model: { + type: "string", + valueHint: "", + description: + "Model ID (default: fun-asr). Async: fun-asr / *-filetrans / paraformer-*; sync: qwen3-asr-flash* / fun-asr-flash* / qwen-audio-*-asr-flash", + }, + language: { + type: "string", + valueHint: "", + description: + "Language hint (e.g. zh, en, ja). Classic async/input-audio: language_hints; qwen3-filetrans: language; qwen3 sync: asr_options.language", + }, diarization: { type: "switch", description: "Enable automatic speaker diarization" }, speakerCount: { type: "number", @@ -55,8 +72,33 @@ const RECOGNIZE_FLAGS = { } satisfies FlagsDef; type RecognizeFlags = ParsedFlags; +function assertSyncFlashFlagsAllowed( + flags: RecognizeFlags, + model: string, + flashFamily: AsrFlashFamily, +): void { + const unsupported: string[] = []; + if (flags.diarization === true) unsupported.push("--diarization"); + if (flags.speakerCount !== undefined) unsupported.push("--speaker-count"); + // qwen3 sync Flash does not use vocabulary_id; input-audio Flash (fun-asr-flash* / qwen-audio-*-asr-flash) does + if (flashFamily === "qwen3" && flags.vocabularyId !== undefined) { + unsupported.push("--vocabulary-id"); + } + if (flags.channelId !== undefined) unsupported.push("--channel-id"); + if (flags.async === true) unsupported.push("--async"); + if (flags.pollInterval !== undefined) unsupported.push("--poll-interval"); + + if (unsupported.length > 0) { + throw new BailianError( + `Model "${model}" uses sync Flash ASR and does not support: ${unsupported.join(", ")}.\n` + + `Hint: Use an async filetrans model (e.g. fun-asr, qwen3-asr-flash-filetrans) for those flags.`, + ExitCode.USAGE, + ); + } +} + export default defineCommand({ - description: "Recognize speech from audio files (FunAudio-ASR)", + description: "Recognize speech from audio files (FunAudio-ASR / Qwen-ASR Flash)", auth: "apiKey", usageArgs: "--url [flags]", flags: RECOGNIZE_FLAGS, @@ -68,6 +110,7 @@ export default defineCommand({ "--url https://example.com/audio.mp3 --vocabulary-id vocab-abc123", "--url https://example.com/audio.mp3 --out result.json", "--url https://example.com/audio.mp3 --async --quiet", + "--url https://example.com/audio.mp3 --model qwen-audio-3.0-asr-flash --language en", ], async run(ctx) { const { settings, flags } = ctx; @@ -90,22 +133,70 @@ export default defineCommand({ } const model = flags.model || "fun-asr"; + const route = resolveAsrApi(model); + if (route.kind === "unsupported") { + throw new BailianError( + route.unsupportedReason ?? `Unsupported ASR model: ${model}`, + ExitCode.USAGE, + ); + } + + if (route.kind === "sync-flash") { + assertSyncFlashFlagsAllowed(flags, model, route.flashFamily!); + if (rawUrls.length !== 1) { + throw new BailianError( + `Model "${model}" is a sync Flash ASR model and accepts exactly one --url (got ${rawUrls.length}).\n` + + `Hint: Pass a single audio URL, or use an async filetrans model for batch files.`, + ExitCode.USAGE, + ); + } + } + if ( + route.kind === "async-filetrans" && + route.asyncInputStyle === "file_url" && + rawUrls.length !== 1 + ) { + throw new BailianError( + `Model "${model}" accepts exactly one --url (got ${rawUrls.length}).\n` + + "Hint: qwen3-asr-flash-filetrans* requires a single file_url.", + ExitCode.USAGE, + ); + } + const format = detectOutputFormat(settings.output); // Auto-upload local files in parallel - const resolvedUrls = await Promise.all(rawUrls.map((u) => ctx.client.uploadFile(u, model))); + const resolvedUrls = await Promise.all(rawUrls.map((url) => ctx.client.uploadFile(url, model))); + + if (route.kind === "sync-flash") { + await handleSyncFlashMode( + ctx.client, + settings, + flags, + format, + model, + route, + resolvedUrls[0]!, + ); + return; + } + const channelId = flags.channelId; - const language = flags.language; const vocabularyId = flags.vocabularyId; + const languageFields = buildAsyncAsrLanguageFields( + route.asyncLanguageStyle ?? "language_hints", + flags.language, + ); const body: DashScopeASRRequest = { model, - input: { - file_urls: resolvedUrls, - }, + input: + route.asyncInputStyle === "file_url" + ? { file_url: resolvedUrls[0]! } + : { file_urls: resolvedUrls }, parameters: { channel_id: channelId !== undefined ? [channelId] : [0], - language_hints: language ? [language] : undefined, + ...languageFields, diarization_enabled: diarization ? true : undefined, speaker_count: speakerCount, vocabulary_id: vocabularyId, @@ -116,7 +207,7 @@ export default defineCommand({ stripUndefined(body.parameters as Record); if (settings.dryRun) { - emitResult({ request: body, mode: "async" }, format); + emitResult({ request: body, mode: "async", path: speechRecognizePath() }, format); return; } @@ -128,6 +219,55 @@ export default defineCommand({ }, }); +async function handleSyncFlashMode( + client: Client, + settings: Settings, + flags: RecognizeFlags, + format: OutputFormat, + model: string, + route: AsrApiRoute, + audioUrl: string, +): Promise { + const flashFamily = route.flashFamily as AsrFlashFamily; + const body = buildAsrFlashRequest({ + model, + audioUrl, + language: flags.language, + vocabularyId: flags.vocabularyId, + flashFamily, + }); + + if (settings.dryRun) { + emitResult({ request: body, mode: "sync", path: route.path }, format); + return; + } + + if (!settings.quiet) { + process.stderr.write(`[Model: ${model}] [Mode: sync] [Files: 1]\n`); + } + + const response = await client.requestJson>({ + path: route.path, + method: "POST", + headers: { "X-DashScope-SSE": "disable" }, + body, + }); + + const text = extractAsrFlashText(response, flashFamily); + if (text) { + process.stdout.write(text.endsWith("\n") ? text : `${text}\n`); + } else { + emitBare(JSON.stringify(response)); + } + + if (flags.out) { + writeFileSync(flags.out, JSON.stringify(response, null, 2) + "\n"); + if (!settings.quiet) { + process.stderr.write(`Full result saved to: ${flags.out}\n`); + } + } +} + async function handleAsyncMode( client: Client, settings: Settings, @@ -160,16 +300,16 @@ async function handleAsyncMode( url: pollUrl, intervalSec: pollInterval, timeoutSec: settings.timeout, - isComplete: (d) => (d as DashScopeASRTaskResult).output.task_status === "SUCCEEDED", - isFailed: (d) => (d as DashScopeASRTaskResult).output.task_status === "FAILED", - getStatus: (d) => (d as DashScopeASRTaskResult).output.task_status, - getErrorMessage: (d) => { - const o = (d as DashScopeASRTaskResult).output; - return (o as unknown as Record).message as string | undefined; + isComplete: (data) => (data as DashScopeASRTaskResult).output.task_status === "SUCCEEDED", + isFailed: (data) => (data as DashScopeASRTaskResult).output.task_status === "FAILED", + getStatus: (data) => (data as DashScopeASRTaskResult).output.task_status, + getErrorMessage: (data) => { + const output = (data as DashScopeASRTaskResult).output; + return (output as unknown as Record).message as string | undefined; }, }); - const results = result.output.results ?? []; + const results = collectAsrTranscriptionItems(result.output); if (results.length === 0) { emitResult({ task_id: taskId, status: result.output.task_status }, format); @@ -179,12 +319,14 @@ async function handleAsyncMode( // Collect all transcription data for --out const allTransData: Record[] = []; - for (let i = 0; i < results.length; i++) { - const subResult = results[i]!; + for (let index = 0; index < results.length; index++) { + const subResult = results[index]!; const isMulti = fileCount > 1; if (isMulti) { - process.stdout.write(`=== [${i + 1}/${results.length}] ${subResult.file_url ?? ""} ===\n`); + process.stdout.write( + `=== [${index + 1}/${results.length}] ${subResult.file_url ?? ""} ===\n`, + ); } if (subResult.subtask_status === "FAILED") { diff --git a/packages/commands/src/commands/usage/shared.ts b/packages/commands/src/commands/usage/shared.ts index 93e32c2..04e4149 100644 --- a/packages/commands/src/commands/usage/shared.ts +++ b/packages/commands/src/commands/usage/shared.ts @@ -24,6 +24,14 @@ export function formatDate(ts: number): string { return `${year}-${month}-${day}`; } +export function formatDateTime(ts: number): string { + const date = new Date(ts); + const hour = String(date.getHours()).padStart(2, "0"); + const minute = String(date.getMinutes()).padStart(2, "0"); + const second = String(date.getSeconds()).padStart(2, "0"); + return `${formatDate(ts)} ${hour}:${minute}:${second}`; +} + export function requireWorkspaceId(settings: Settings, binName: string): string { if (settings.workspaceId) return settings.workspaceId; diff --git a/packages/commands/src/commands/usage/token-plan.ts b/packages/commands/src/commands/usage/token-plan.ts new file mode 100644 index 0000000..d4bb8bf --- /dev/null +++ b/packages/commands/src/commands/usage/token-plan.ts @@ -0,0 +1,146 @@ +import { defineCommand, detectOutputFormat, unwrapResponse } from "bailian-cli-core"; +import { + ansi, + displayWidth, + emitResult, + type AnsiStyles, + type TextStyle, +} from "bailian-cli-runtime"; +import { formatDateTime } from "./shared.ts"; + +const TOKEN_PLAN_USAGE_API = "zeldaHttp.apikeyMgr./tokenplan/personal/api/v2/usage"; +const BOX_WIDTH = 76; +const PROGRESS_WIDTH = 32; + +interface TokenPlanUsage { + per5HourPercentage?: number; + per5HourResetTime?: number; + per1WeekPercentage?: number; + per1WeekResetTime?: number; +} + +interface QuotaWindow { + percentage?: number; + resetTime?: number; +} + +/** Accept only finite numbers; anything else counts as absent (possibly unlimited). */ +function readNumber(value: unknown): number | undefined { + return typeof value === "number" && Number.isFinite(value) ? value : undefined; +} + +function readUsage(result: unknown): TokenPlanUsage { + const response = unwrapResponse(result as Record); + const usage: TokenPlanUsage = {}; + + const per5HourPercentage = readNumber(response.per5HourPercentage); + if (per5HourPercentage !== undefined) usage.per5HourPercentage = per5HourPercentage; + const per5HourResetTime = readNumber(response.per5HourResetTime); + if (per5HourResetTime !== undefined) usage.per5HourResetTime = per5HourResetTime; + const per1WeekPercentage = readNumber(response.per1WeekPercentage); + if (per1WeekPercentage !== undefined) usage.per1WeekPercentage = per1WeekPercentage; + const per1WeekResetTime = readNumber(response.per1WeekResetTime); + if (per1WeekResetTime !== undefined) usage.per1WeekResetTime = per1WeekResetTime; + + return usage; +} + +function formatPercentage(ratio: number): string { + return `${(ratio * 100).toFixed(2)}%`; +} + +function formatRemainingTime(resetTime: number, now: number): string { + const remainingMs = Math.max(0, resetTime - now); + const totalMinutes = Math.floor(remainingMs / 60_000); + if (totalMinutes === 0) return "now"; + + const days = Math.floor(totalMinutes / (24 * 60)); + const hours = Math.floor((totalMinutes % (24 * 60)) / 60); + const minutes = totalMinutes % 60; + const parts: string[] = []; + if (days > 0) parts.push(`${days}d`); + if (hours > 0) parts.push(`${hours}h`); + if (minutes > 0 || parts.length === 0) parts.push(`${minutes}m`); + return parts.join(" "); +} + +function progressBar(ratio: number): string { + const clampedRatio = Math.min(1, Math.max(0, ratio)); + const filled = Math.round(clampedRatio * PROGRESS_WIDTH); + return `[${"█".repeat(filled)}${"░".repeat(PROGRESS_WIDTH - filled)}]`; +} + +function progressStyle(percentage: number, color: AnsiStyles): TextStyle { + if (percentage >= 0.9) return color.red; + if (percentage >= 0.75) return color.yellow; + return color.green; +} + +function printView(usage: TokenPlanUsage, generatedAt: number): void { + const color = ansi(process.stdout); + const writeLine = (text = "", style?: TextStyle) => { + const padding = Math.max(0, BOX_WIDTH - displayWidth(` ${text}`)); + process.stdout.write(`│ ${style ? style(text) : text}${" ".repeat(padding)}│\n`); + }; + const writeQuota = (label: string, unlimitedMessage: string, window: QuotaWindow) => { + writeLine(label, color.bold); + if (window.percentage === undefined) { + writeLine(unlimitedMessage, color.dim); + return; + } + + const percentageText = formatPercentage(window.percentage); + const bar = progressBar(window.percentage); + writeLine(`${percentageText} used ${bar}`, progressStyle(window.percentage, color)); + if (window.resetTime === undefined) { + writeLine("Resets: not applicable (no usage yet)", color.dim); + return; + } + + const resetText = `Resets: ${formatDateTime(window.resetTime)} (in ${formatRemainingTime(window.resetTime, generatedAt)})`; + writeLine(resetText, color.dim); + }; + + process.stdout.write(`┌${"─".repeat(BOX_WIDTH)}┐\n`); + writeLine("Token Plan Usage", color.cyan); + writeLine(`Generated at: ${formatDateTime(generatedAt)} (local time)`, color.dim); + process.stdout.write(`├${"─".repeat(BOX_WIDTH)}┤\n`); + writeQuota( + "5-hour quota", + "The 5-hour limit may be unlimited; verify in the Bailian Token Plan console.", + { percentage: usage.per5HourPercentage, resetTime: usage.per5HourResetTime }, + ); + process.stdout.write(`├${"─".repeat(BOX_WIDTH)}┤\n`); + writeQuota( + "1-week quota", + "The 1-week limit may be unlimited; verify in the Bailian Token Plan console.", + { percentage: usage.per1WeekPercentage, resetTime: usage.per1WeekResetTime }, + ); + process.stdout.write(`└${"─".repeat(BOX_WIDTH)}┘\n`); +} + +export default defineCommand({ + description: "Show Token Plan quota usage", + auth: "console", + usageArgs: "[flags]", + exampleArgs: ["", "--output json"], + async run(ctx) { + const { settings } = ctx; + const format = detectOutputFormat(settings.output); + + if (settings.dryRun) { + emitResult({ api: TOKEN_PLAN_USAGE_API, data: {} }, format); + return; + } + + const result = await ctx.client.console(TOKEN_PLAN_USAGE_API, {}); + const usage = readUsage(result); + + if (format === "json") { + emitResult(usage, format); + return; + } + + printView(usage, Date.now()); + }, +}); diff --git a/packages/commands/src/index.ts b/packages/commands/src/index.ts index a5af82d..a0761a3 100644 --- a/packages/commands/src/index.ts +++ b/packages/commands/src/index.ts @@ -48,6 +48,7 @@ export { default as usageFree } from "./commands/usage/free.ts"; export { default as usageFreetier } from "./commands/usage/freetier.ts"; export { default as usageStats } from "./commands/usage/stats.ts"; export { default as usageSummary } from "./commands/usage/summary.ts"; +export { default as usageTokenPlan } from "./commands/usage/token-plan.ts"; export { default as pipelineRun } from "./commands/pipeline/run.ts"; export { default as pipelineValidate } from "./commands/pipeline/validate.ts"; export { default as advisorRecommend } from "./commands/advisor/recommend.ts"; diff --git a/packages/commands/tests/e2e/speech-recognize.e2e.test.ts b/packages/commands/tests/e2e/speech-recognize.e2e.test.ts index 926af0e..2805f16 100644 --- a/packages/commands/tests/e2e/speech-recognize.e2e.test.ts +++ b/packages/commands/tests/e2e/speech-recognize.e2e.test.ts @@ -1,4 +1,6 @@ import { readFileSync } from "node:fs"; +import http from "node:http"; +import type { AddressInfo } from "node:net"; import { join } from "node:path"; import { describe, expect, test } from "vite-plus/test"; import { @@ -16,6 +18,37 @@ import { SPEECH_ROUTES } from "./topic-routes.ts"; */ describe("e2e: speech recognize", () => { + async function runRecognizeDryRun(args: string[]) { + const { stdout, stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [ + "speech", + "recognize", + ...args, + "--dry-run", + "--output", + "json", + "--quiet", + ]); + expect(exitCode, stderr).toBe(0); + return parseStdoutJson<{ + mode?: string; + path?: string; + request?: { + model?: string; + parameters?: { + format?: string; + language_hints?: string[]; + language?: string; + vocabulary_id?: string; + }; + input?: { + file_url?: string; + file_urls?: string[]; + messages?: Array<{ content?: Array<{ type?: string }> }>; + }; + }; + }>(stdout); + } + test("speech recognize --help 正常退出", async () => { const { stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [ "speech", @@ -25,6 +58,217 @@ describe("e2e: speech recognize", () => { expect(exitCode, stderr).toBe(0); expect(stderr).toMatch(/recognize|--url|model|audio/i); }); + + test("speech recognize sync-flash dry-run 走 multimodal-generation", async () => { + const body = await runRecognizeDryRun([ + "--model", + "qwen-audio-3.0-asr-flash", + "--url", + "https://dashscope.oss-cn-beijing.aliyuncs.com/samples/audio/paraformer/hello_world_female2.wav", + "--language", + "en", + "--vocabulary-id", + "vocab-e2e", + ]); + expect(body.mode).toBe("sync"); + expect(body.path).toBe("/api/v1/services/aigc/multimodal-generation/generation"); + expect(body.request?.model).toBe("qwen-audio-3.0-asr-flash"); + expect(body.request?.parameters?.format).toBe("wav"); + expect(body.request?.parameters?.language_hints).toEqual(["en"]); + expect(body.request?.parameters?.vocabulary_id).toBe("vocab-e2e"); + expect(body.request?.input?.messages?.[0]?.content?.[0]?.type).toBe("input_audio"); + }); + + test("speech recognize qwen3 filetrans dry-run 使用 file_url 与 language", async () => { + const body = await runRecognizeDryRun([ + "--model", + "qwen3-asr-flash-filetrans", + "--url", + "https://dashscope.oss-cn-beijing.aliyuncs.com/samples/audio/paraformer/hello_world_female2.wav", + "--language", + "zh", + ]); + expect(body.mode).toBe("async"); + expect(body.path).toBe("/api/v1/services/audio/asr/transcription"); + expect(body.request?.input?.file_url?.startsWith("https://")).toBe(true); + expect(body.request?.input?.file_urls).toBeUndefined(); + expect(body.request?.parameters?.language).toBe("zh"); + expect(body.request?.parameters?.language_hints).toBeUndefined(); + }); + + test("speech recognize realtime 模型报用法错误", async () => { + // Use --dry-run to skip auth so CI without API keys still hits USAGE(2) + const { stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [ + "speech", + "recognize", + "--model", + "qwen3-asr-flash-realtime", + "--url", + "https://example.com/a.wav", + "--dry-run", + "--quiet", + ]); + expect(exitCode).toBe(2); + expect(stderr).toMatch(/realtime|WebSocket|unsupported/i); + }); + + test("speech recognize flash 真实请求走 sync endpoint 并落盘 --out", async () => { + let requestPath = ""; + let requestBody: Record = {}; + let sseHeader: string | undefined; + const server = http.createServer((request, response) => { + const chunks: Buffer[] = []; + request.on("data", (chunk: Buffer) => chunks.push(chunk)); + request.on("end", () => { + requestPath = request.url ?? ""; + requestBody = JSON.parse(Buffer.concat(chunks).toString("utf8")) as Record; + sseHeader = request.headers["x-dashscope-sse"] as string | undefined; + response.writeHead(200, { "Content-Type": "application/json" }); + response.end( + JSON.stringify({ + output: { text: "flash recognition works" }, + request_id: "request-146", + }), + ); + }); + }); + await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); + const address = server.address() as AddressInfo; + const outDir = makeE2eOutputDir("speech-recognize-flash-sync"); + const outPath = join(outDir, "result.json"); + + try { + const { stdout, stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [ + "speech", + "recognize", + "--model", + "fun-asr-flash-2026-06-15", + "--url", + "https://example.com/sample.wav", + "--api-key", + "sk-e2e-placeholder", + "--base-url", + `http://127.0.0.1:${address.port}`, + "--out", + outPath, + "--quiet", + ]); + + expect(exitCode, stderr).toBe(0); + expect(stdout).toContain("flash recognition works"); + expect(requestPath).toBe("/api/v1/services/aigc/multimodal-generation/generation"); + expect(sseHeader).toBe("disable"); + expect(requestBody).toMatchObject({ + model: "fun-asr-flash-2026-06-15", + parameters: { format: "wav" }, + }); + expect(JSON.parse(readFileSync(outPath, "utf8"))).toMatchObject({ + output: { text: "flash recognition works" }, + request_id: "request-146", + }); + } finally { + await new Promise((resolve) => server.close(() => resolve())); + } + }); + + test("speech recognize flash 多 --url 在发请求前报用法错误", async () => { + const { stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [ + "speech", + "recognize", + "--model", + "qwen-audio-3.0-asr-flash", + "--url", + "https://example.com/a.wav", + "--url", + "https://example.com/b.wav", + "--dry-run", + "--quiet", + ]); + + expect(exitCode).toBe(2); + expect(stderr).toMatch(/exactly one --url|sync Flash/i); + }); + + test("speech recognize qwen3-filetrans 轮询成功后下载 result.transcription_url", async () => { + const server = http.createServer((request, response) => { + const url = request.url ?? ""; + const chunks: Buffer[] = []; + request.on("data", (chunk: Buffer) => chunks.push(chunk)); + request.on("end", () => { + response.writeHead(200, { "Content-Type": "application/json" }); + if (url.startsWith("/api/v1/services/audio/asr/transcription")) { + response.end( + JSON.stringify({ + output: { task_id: "task-qwen3", task_status: "PENDING" }, + request_id: "req-submit", + }), + ); + return; + } + if (url.startsWith("/api/v1/tasks/")) { + const address = server.address() as AddressInfo; + response.end( + JSON.stringify({ + output: { + task_id: "task-qwen3", + task_status: "SUCCEEDED", + result: { + transcription_url: `http://127.0.0.1:${address.port}/transcription.json`, + }, + }, + request_id: "req-poll", + }), + ); + return; + } + if (url.startsWith("/transcription.json")) { + response.end( + JSON.stringify({ + file_url: "https://example.com/a.wav", + transcripts: [{ text: "你好世界", sentences: [{ text: "你好世界" }] }], + }), + ); + return; + } + response.writeHead(404); + response.end(JSON.stringify({ message: `unexpected path: ${url}` })); + }); + }); + await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); + const address = server.address() as AddressInfo; + const outDir = makeE2eOutputDir("speech-recognize-qwen3-filetrans"); + const outPath = join(outDir, "result.json"); + + try { + const { stdout, stderr, exitCode } = await runCommandE2e(SPEECH_ROUTES, [ + "speech", + "recognize", + "--model", + "qwen3-asr-flash-filetrans", + "--url", + "https://example.com/a.wav", + "--language", + "zh", + "--api-key", + "sk-e2e-placeholder", + "--base-url", + `http://127.0.0.1:${address.port}`, + "--poll-interval", + "1", + "--out", + outPath, + "--quiet", + ]); + + expect(exitCode, stderr).toBe(0); + expect(stdout).toContain("你好世界"); + expect(JSON.parse(readFileSync(outPath, "utf8"))).toMatchObject({ + transcripts: [{ text: "你好世界" }], + }); + } finally { + await new Promise((resolve) => server.close(() => resolve())); + } + }); }); describe.skipIf(!isBailianE2EMediaEnabled() || !isDashScopeE2EReady())( diff --git a/packages/commands/tests/e2e/topic-routes.ts b/packages/commands/tests/e2e/topic-routes.ts index 84761f4..5452402 100644 --- a/packages/commands/tests/e2e/topic-routes.ts +++ b/packages/commands/tests/e2e/topic-routes.ts @@ -109,6 +109,7 @@ export const USAGE_ROUTES: E2eRouteExports = { "usage free": "usageFree", "usage freetier": "usageFreetier", "usage stats": "usageStats", + "usage token-plan": "usageTokenPlan", }; export const DEPLOY_ROUTES: E2eRouteExports = { diff --git a/packages/commands/tests/e2e/usage-token-plan.e2e.test.ts b/packages/commands/tests/e2e/usage-token-plan.e2e.test.ts new file mode 100644 index 0000000..1adc592 --- /dev/null +++ b/packages/commands/tests/e2e/usage-token-plan.e2e.test.ts @@ -0,0 +1,76 @@ +import { describe, expect, test } from "vite-plus/test"; +import { + isConsoleAuthFailure, + isConsoleE2EReady, + parseStdoutJson, + runCommandE2e, +} from "./helpers.ts"; +import { USAGE_ROUTES } from "./topic-routes.ts"; + +describe("e2e: usage token-plan", () => { + test("usage token-plan --help 正常退出", async () => { + const { stderr, exitCode } = await runCommandE2e(USAGE_ROUTES, [ + "usage", + "token-plan", + "--help", + ]); + expect(exitCode, stderr).toBe(0); + expect(stderr).toMatch(/Token Plan|quota/i); + }); + + test("usage token-plan --help 包含 --output json 示例", async () => { + const { stderr, exitCode } = await runCommandE2e(USAGE_ROUTES, [ + "usage", + "token-plan", + "--help", + ]); + expect(exitCode, stderr).toBe(0); + expect(stderr).toContain("bl usage token-plan --output json"); + }); +}); + +describe.skipIf(!isConsoleE2EReady())("e2e: usage token-plan(Console)", () => { + test("usage token-plan --dry-run 输出网关请求计划", async () => { + const { stdout, stderr, exitCode } = await runCommandE2e(USAGE_ROUTES, [ + "usage", + "token-plan", + "--dry-run", + "--output", + "json", + ]); + expect(exitCode, stderr).toBe(0); + const data = parseStdoutJson<{ api?: string; data?: Record }>(stdout); + expect(data.api).toBe("zeldaHttp.apikeyMgr./tokenplan/personal/api/v2/usage"); + expect(data.data).toEqual({}); + }); + + test("usage token-plan --output json 返回可用的额度字段", async () => { + const result = await runCommandE2e(USAGE_ROUTES, ["usage", "token-plan", "--output", "json"]); + if (isConsoleAuthFailure(result)) return; + expect(result.exitCode, result.stderr).toBe(0); + const data = parseStdoutJson<{ + per5HourPercentage?: number; + per5HourResetTime?: number; + per1WeekPercentage?: number; + per1WeekResetTime?: number; + }>(result.stdout); + const fields = [ + data.per5HourPercentage, + data.per5HourResetTime, + data.per1WeekPercentage, + data.per1WeekResetTime, + ]; + for (const field of fields) { + if (field !== undefined) expect(field).toBeTypeOf("number"); + } + }); + + test("usage token-plan 默认渲染生成时间与两个额度窗口", async () => { + const result = await runCommandE2e(USAGE_ROUTES, ["usage", "token-plan"]); + if (isConsoleAuthFailure(result)) return; + expect(result.exitCode, result.stderr).toBe(0); + expect(result.stdout).toContain("Generated at:"); + expect(result.stdout).toContain("5-hour quota"); + expect(result.stdout).toContain("1-week quota"); + }); +}); diff --git a/packages/commands/tests/token-plan-usage.test.ts b/packages/commands/tests/token-plan-usage.test.ts new file mode 100644 index 0000000..fe551f5 --- /dev/null +++ b/packages/commands/tests/token-plan-usage.test.ts @@ -0,0 +1,182 @@ +import { afterEach, describe, expect, test, vi } from "vite-plus/test"; +import tokenPlanUsage from "../src/commands/usage/token-plan.ts"; + +const originalNoColor = process.env.NO_COLOR; +const originalIsTty = Object.getOwnPropertyDescriptor(process.stdout, "isTTY"); + +afterEach(() => { + if (originalNoColor === undefined) delete process.env.NO_COLOR; + else process.env.NO_COLOR = originalNoColor; + if (originalIsTty) Object.defineProperty(process.stdout, "isTTY", originalIsTty); + else delete (process.stdout as { isTTY?: boolean }).isTTY; + vi.restoreAllMocks(); +}); + +function captureStdout(): string[] { + const output: string[] = []; + vi.spyOn(process.stdout, "write").mockImplementation((chunk) => { + output.push(String(chunk)); + return true; + }); + return output; +} + +async function runTokenPlan(response: Record, output?: string): Promise { + await tokenPlanUsage.run({ + client: { console: vi.fn().mockResolvedValue(response) }, + flags: {}, + settings: { dryRun: false, output }, + } as never); +} + +function makeUsageResponse( + per5HourPercentage?: number, + per1WeekPercentage = per5HourPercentage, +): Record { + const usage: Record = {}; + if (per5HourPercentage !== undefined) { + usage.per5HourPercentage = per5HourPercentage; + if (per5HourPercentage !== 0) usage.per5HourResetTime = 1_786_000_000_000; + } + if (per1WeekPercentage !== undefined) { + usage.per1WeekPercentage = per1WeekPercentage; + if (per1WeekPercentage !== 0) usage.per1WeekResetTime = 1_786_100_000_000; + } + + return wrapResponse(usage); +} + +function wrapResponse(usage: Record): Record { + return { + data: { + DataV2: { + data: { + data: usage, + }, + }, + }, + }; +} + +describe("usage token-plan view", () => { + test.each([ + [0.7499, "32"], + [0.75, "33"], + [0.9, "31"], + ])("uses ANSI color %s for %s", async (percentage, colorCode) => { + delete process.env.NO_COLOR; + Object.defineProperty(process.stdout, "isTTY", { configurable: true, value: true }); + const output = captureStdout(); + + await runTokenPlan(makeUsageResponse(percentage)); + + expect(output.join("")).toContain(`\u001B[${colorCode}m`); + }); + + test("accepts missing reset times when the quota usage is zero", async () => { + const output = captureStdout(); + + await runTokenPlan(makeUsageResponse(0)); + + expect(output.join("")).toContain("Resets: not applicable (no usage yet)"); + }); + + test("allows one unused quota window without masking another reset time", async () => { + const output = captureStdout(); + + await runTokenPlan(makeUsageResponse(0, 0.5)); + + const renderedOutput = output.join(""); + expect(renderedOutput).toContain("Resets: not applicable (no usage yet)"); + expect(renderedOutput).toMatch(/Resets: \d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2}/); + }); + + test("renders missing quota windows as possibly unlimited", async () => { + const output = captureStdout(); + + await runTokenPlan(makeUsageResponse()); + + const renderedOutput = output.join(""); + expect(renderedOutput).toContain( + "The 5-hour limit may be unlimited; verify in the Bailian Token Plan console.", + ); + expect(renderedOutput).toContain( + "The 1-week limit may be unlimited; verify in the Bailian Token Plan console.", + ); + }); + + test("renders only the missing quota window as possibly unlimited", async () => { + const output = captureStdout(); + + await runTokenPlan(makeUsageResponse(undefined, 0.5)); + + const renderedOutput = output.join(""); + expect(renderedOutput).toContain( + "The 5-hour limit may be unlimited; verify in the Bailian Token Plan console.", + ); + expect(renderedOutput).not.toContain( + "The 1-week limit may be unlimited; verify in the Bailian Token Plan console.", + ); + expect(renderedOutput).toMatch(/Resets: \d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2}/); + }); + + test("renders a window with a missing percentage as possibly unlimited even when its reset time is present", async () => { + const output = captureStdout(); + + await runTokenPlan(wrapResponse({ per5HourResetTime: 1_786_000_000_000 })); + + expect(output.join("")).toContain( + "The 5-hour limit may be unlimited; verify in the Bailian Token Plan console.", + ); + }); + + test("treats non-numeric quota fields as absent instead of failing", async () => { + const output = captureStdout(); + + await runTokenPlan( + wrapResponse({ per5HourPercentage: "not-a-number", per1WeekPercentage: Number.NaN }), + ); + + const renderedOutput = output.join(""); + expect(renderedOutput).toContain( + "The 5-hour limit may be unlimited; verify in the Bailian Token Plan console.", + ); + expect(renderedOutput).toContain( + "The 1-week limit may be unlimited; verify in the Bailian Token Plan console.", + ); + }); +}); + +describe("usage token-plan json", () => { + test("outputs the four core usage fields with --output json", async () => { + const output = captureStdout(); + + await runTokenPlan(makeUsageResponse(0.5, 0.25), "json"); + + expect(JSON.parse(output.join(""))).toEqual({ + per5HourPercentage: 0.5, + per5HourResetTime: 1_786_000_000_000, + per1WeekPercentage: 0.25, + per1WeekResetTime: 1_786_100_000_000, + }); + }); + + test("returns an empty JSON object when no quota fields are available", async () => { + const output = captureStdout(); + + await runTokenPlan(makeUsageResponse(), "json"); + + expect(output.join("").trim()).toBe("{}"); + }); + + test("omits non-numeric quota fields from the JSON output", async () => { + const output = captureStdout(); + + await runTokenPlan( + wrapResponse({ per5HourPercentage: "not-a-number", per1WeekPercentage: 0 }), + "json", + ); + + expect(JSON.parse(output.join(""))).toEqual({ per1WeekPercentage: 0 }); + }); +}); diff --git a/packages/core/src/client/asr-routes.ts b/packages/core/src/client/asr-routes.ts new file mode 100644 index 0000000..0d982f1 --- /dev/null +++ b/packages/core/src/client/asr-routes.ts @@ -0,0 +1,327 @@ +import { imageSyncPath, speechRecognizePath } from "./endpoints.ts"; + +/** + * DashScope ASR APIs differ by model family: + * + * - async file transcription (`.../audio/asr/transcription`): + * fun-asr*, paraformer* (non-realtime), *-filetrans, sensevoice* + * language via `parameters.language_hints` + * - sync multimodal (`.../aigc/multimodal-generation/generation`): + * - qwen3: `{ content: [{ audio }] }` + optional `asr_options.language` + * (qwen3-asr-flash*) + * - input-audio: `{ type: input_audio, input_audio.data }` + + * `format`/`sample_rate` + optional `language_hints` + * (fun-asr-flash*, qwen-audio-*-asr-flash*) + * - realtime / streaming: WebSocket — not supported by `speech recognize` + */ + +export type AsrApiKind = "async-filetrans" | "sync-flash" | "unsupported"; + +/** Sync-flash request body shape differs by Flash protocol family. */ +export type AsrFlashFamily = "qwen3" | "input-audio"; + +export interface AsrApiRoute { + kind: AsrApiKind; + path: string; + /** True when the call is synchronous (no X-DashScope-Async / task poll). */ + useSync: boolean; + /** + * Async transcription request input style. + * - `file_urls`: classic async models (fun-asr / paraformer / qwen-audio filetrans...) + * - `file_url`: qwen3-asr-flash-filetrans family + */ + asyncInputStyle?: "file_urls" | "file_url"; + /** + * Async transcription language field style. + * - `language_hints`: fun-asr / paraformer / qwen-audio filetrans... + * - `language`: qwen3-asr-flash-filetrans* + */ + asyncLanguageStyle?: "language_hints" | "language"; + flashFamily?: AsrFlashFamily; + /** Human-readable reason when kind is unsupported. */ + unsupportedReason?: string; +} + +function isRealtimeOrStreaming(model: string): boolean { + return /realtime|streaming/i.test(model); +} + +function isFiletransModel(model: string): boolean { + return /filetrans/i.test(model); +} + +function isQwen3FiletransModel(model: string): boolean { + return /^qwen3-asr-flash-filetrans(?:-|$)/i.test(model); +} + +const INPUT_AUDIO_FLASH_PREFIXES = ["fun-asr-flash", "qwen-audio"] as const; + +/** + * Fun-ASR-Flash / Qwen-Audio-*-ASR-Flash share the input_audio + format protocol. + * Examples: fun-asr-flash-2026-06-15, qwen-audio-3.0-asr-flash + */ +function isInputAudioFlashModel(model: string): boolean { + if (isRealtimeOrStreaming(model) || isFiletransModel(model)) return false; + if (model.startsWith(INPUT_AUDIO_FLASH_PREFIXES[0])) return true; + if (model.startsWith(INPUT_AUDIO_FLASH_PREFIXES[1]) && /asr-flash/i.test(model)) return true; + return false; +} + +/** + * Qwen3-ASR-Flash sync models use content.audio + asr_options. + * Examples: qwen3-asr-flash, qwen3-asr-flash-2025-09-08, qwen3-asr-flash-us + */ +function isQwen3AsrFlashModel(model: string): boolean { + if (!/^qwen3-asr-flash(?:-|$)/i.test(model)) return false; + if (isFiletransModel(model) || isRealtimeOrStreaming(model)) return false; + if (isInputAudioFlashModel(model)) return false; + return true; +} + +/** + * Resolve which DashScope ASR API a model should use for file recognition. + * Unknown models default to async-filetrans (preserves existing CLI behavior). + */ +export function resolveAsrApi(model: string): AsrApiRoute { + if (isRealtimeOrStreaming(model)) { + return { + kind: "unsupported", + path: "", + useSync: false, + unsupportedReason: + `Model "${model}" is a realtime/streaming ASR model and requires a WebSocket API. ` + + `Use an async filetrans model (e.g. fun-asr, qwen3-asr-flash-filetrans) or a sync flash model ` + + `(e.g. qwen3-asr-flash, qwen-audio-3.0-asr-flash) with this command.`, + }; + } + + if (isFiletransModel(model)) { + const isQwen3Filetrans = isQwen3FiletransModel(model); + return { + kind: "async-filetrans", + path: speechRecognizePath(), + useSync: false, + asyncInputStyle: isQwen3Filetrans ? "file_url" : "file_urls", + asyncLanguageStyle: isQwen3Filetrans ? "language" : "language_hints", + }; + } + + if (isInputAudioFlashModel(model)) { + return { + kind: "sync-flash", + path: imageSyncPath(), + useSync: true, + flashFamily: "input-audio", + }; + } + + if (isQwen3AsrFlashModel(model)) { + return { + kind: "sync-flash", + path: imageSyncPath(), + useSync: true, + flashFamily: "qwen3", + }; + } + + // fun-asr / paraformer / sensevoice / unknown → keep legacy async path + return { + kind: "async-filetrans", + path: speechRecognizePath(), + useSync: false, + asyncInputStyle: "file_urls", + asyncLanguageStyle: "language_hints", + }; +} + +/** Infer audio container hint for input-audio Flash `parameters.format`. */ +export function inferAudioFormatHint(audioUrl: string): string { + // data URI: data:audio/mpeg;base64,... → mp3; data:audio/x-wav;... → wav + const dataType = /^data:audio\/([^;,]+)/i.exec(audioUrl)?.[1]?.toLowerCase(); + if (dataType) { + if (dataType === "mpeg") return "mp3"; + if (dataType === "x-wav" || dataType === "wave") return "wav"; + return dataType; + } + + const pathPart = audioUrl.split(/[?#]/, 1)[0] ?? audioUrl; + const match = pathPart.match(/\.([a-zA-Z0-9]+)$/); + const extension = match?.[1]?.toLowerCase(); + if (!extension) return "wav"; + if (extension === "mpeg") return "mp3"; + return extension; +} + +export interface BuildAsrFlashRequestOpts { + model: string; + audioUrl: string; + language?: string; + /** Precompiled hotword vocabulary ID; supported for input-audio Flash (fun-asr-flash* / qwen-audio-*-asr-flash). */ + vocabularyId?: string; + flashFamily: AsrFlashFamily; +} + +/** + * Build language fields for async ASR routes. + * qwen3-asr-flash-filetrans* → `language`; other async models → `language_hints`. + */ +export function buildAsyncAsrLanguageFields( + languageStyle: "language_hints" | "language", + language?: string, +): { language_hints?: string[]; language?: string } { + if (!language) return {}; + if (languageStyle === "language") { + return { language }; + } + return { language_hints: [language] }; +} + +/** Build a sync multimodal ASR request body for Flash models. */ +export function buildAsrFlashRequest(opts: BuildAsrFlashRequestOpts): Record { + const { model, audioUrl, language, vocabularyId, flashFamily } = opts; + + if (flashFamily === "input-audio") { + // Match official Qwen-Audio / Fun-ASR-Flash docs: language_hints + vocabulary_id + const parameters: Record = { + format: inferAudioFormatHint(audioUrl), + sample_rate: "16000", + }; + if (language) { + parameters.language_hints = [language]; + } + if (vocabularyId) { + parameters.vocabulary_id = vocabularyId; + } + return { + model, + input: { + messages: [ + { + role: "user", + content: [ + { + type: "input_audio", + input_audio: { data: audioUrl }, + }, + ], + }, + ], + }, + parameters, + }; + } + + const asrOptions: Record = {}; + if (language) { + asrOptions.language = language; + } + + const parameters: Record = {}; + if (Object.keys(asrOptions).length > 0) { + parameters.asr_options = asrOptions; + } + + const body: Record = { + model, + input: { + messages: [ + { + role: "user", + content: [{ audio: audioUrl }], + }, + ], + }, + }; + if (Object.keys(parameters).length > 0) { + body.parameters = parameters; + } + return body; +} + +/** + * Extract recognition text from a sync Flash ASR response. + * Qwen3 uses choices[].message.content; input-audio Flash uses output.text / + * output.sentence.text / output.output.sentence.text. + */ +export function extractAsrFlashText( + response: Record, + flashFamily: AsrFlashFamily, +): string { + const output = response.output as Record | undefined; + if (!output) return ""; + + if (flashFamily === "input-audio") { + if (typeof output.text === "string" && output.text.length > 0) { + return output.text; + } + const topSentence = output.sentence as Record | undefined; + if (typeof topSentence?.text === "string" && topSentence.text.length > 0) { + return topSentence.text; + } + const nested = output.output as Record | undefined; + const nestedSentence = nested?.sentence as Record | undefined; + if (typeof nestedSentence?.text === "string") { + return nestedSentence.text; + } + return ""; + } + + const choices = output.choices as Array> | undefined; + if (!choices?.length) return ""; + + const texts: string[] = []; + for (const choice of choices) { + const message = choice.message as Record | undefined; + if (!message) continue; + const content = message.content; + if (typeof content === "string") { + texts.push(content); + continue; + } + if (!Array.isArray(content)) continue; + for (const item of content) { + if (typeof item === "string") { + texts.push(item); + continue; + } + if (item && typeof item === "object") { + const record = item as Record; + if (typeof record.text === "string") { + texts.push(record.text); + } + } + } + } + return texts.join(""); +} + +/** + * Normalize async ASR task transcription items: + * - classic models: `output.results[]` + * - qwen3-asr-flash-filetrans*: `output.result.transcription_url` + */ +export function collectAsrTranscriptionItems(output: { + results?: Array<{ + file_url?: string; + transcription_url?: string; + subtask_status?: string; + code?: string; + message?: string; + }>; + result?: { transcription_url?: string }; +}): Array<{ + file_url?: string; + transcription_url?: string; + subtask_status?: string; + code?: string; + message?: string; +}> { + if (output.results && output.results.length > 0) { + return output.results; + } + const transcriptionUrl = output.result?.transcription_url; + if (typeof transcriptionUrl === "string" && transcriptionUrl.length > 0) { + return [{ transcription_url: transcriptionUrl, subtask_status: "SUCCEEDED" }]; + } + return []; +} diff --git a/packages/core/src/client/index.ts b/packages/core/src/client/index.ts index 5884d6a..9fa3d03 100644 --- a/packages/core/src/client/index.ts +++ b/packages/core/src/client/index.ts @@ -34,6 +34,18 @@ export { type ImageInputStyle, type ImageSizeProfile, } from "./image-routes.ts"; +export { + buildAsrFlashRequest, + buildAsyncAsrLanguageFields, + collectAsrTranscriptionItems, + extractAsrFlashText, + inferAudioFormatHint, + resolveAsrApi, + type AsrApiKind, + type AsrApiRoute, + type AsrFlashFamily, + type BuildAsrFlashRequestOpts, +} from "./asr-routes.ts"; export { CHANNEL, sourceConfig, trackingHeaders, type TrackingIdentity } from "./headers.ts"; export type { HttpDeps, RequestOpts } from "./http.ts"; export { request, requestJson } from "./http.ts"; diff --git a/packages/core/src/types/api.ts b/packages/core/src/types/api.ts index c6c00ec..5c10b4a 100644 --- a/packages/core/src/types/api.ts +++ b/packages/core/src/types/api.ts @@ -533,33 +533,46 @@ export interface DashScopeTTSStreamChunk { export interface DashScopeASRRequest { model: string; input: { - file_urls: string[]; + file_urls?: string[]; + file_url?: string; }; parameters?: { channel_id?: number[]; + /** Classic async models (fun-asr / paraformer / qwen-audio filetrans, etc.) */ language_hints?: string[]; + /** qwen3-asr-flash-filetrans* uses singular `language` */ + language?: string; diarization_enabled?: boolean; speaker_count?: number; vocabulary_id?: string; }; } +export interface DashScopeASRTranscriptionItem { + file_url?: string; + transcription_url?: string; + subtask_status?: string; + code?: string; + message?: string; +} + export interface DashScopeASRTaskResult { output: { task_id: string; task_status: "PENDING" | "RUNNING" | "SUCCEEDED" | "FAILED" | "UNKNOWN"; - results?: Array<{ - file_url?: string; + /** Multi-file async results (fun-asr / paraformer / qwen-audio filetrans, etc.) */ + results?: DashScopeASRTranscriptionItem[]; + /** Singular result returned by qwen3-asr-flash-filetrans* on success */ + result?: { transcription_url?: string; - subtask_status?: string; - code?: string; - message?: string; - }>; + }; task_metrics?: { TOTAL: number; SUCCEEDED: number; FAILED: number; }; + code?: string; + message?: string; }; usage?: Record; request_id: string; diff --git a/packages/core/src/types/index.ts b/packages/core/src/types/index.ts index bafd242..bf54779 100644 --- a/packages/core/src/types/index.ts +++ b/packages/core/src/types/index.ts @@ -48,6 +48,7 @@ export type { ChatTool, DashScopeASRRequest, DashScopeASRTaskResult, + DashScopeASRTranscriptionItem, DashScopeAsyncResponse, DashScopeImageRequest, DashScopeImageSyncResponse, diff --git a/packages/core/tests/asr-routes.test.ts b/packages/core/tests/asr-routes.test.ts new file mode 100644 index 0000000..7b27a1e --- /dev/null +++ b/packages/core/tests/asr-routes.test.ts @@ -0,0 +1,213 @@ +import { expect, test } from "vite-plus/test"; +import { + buildAsrFlashRequest, + buildAsyncAsrLanguageFields, + collectAsrTranscriptionItems, + extractAsrFlashText, + inferAudioFormatHint, + resolveAsrApi, +} from "../src/client/asr-routes.ts"; + +test("resolveAsrApi routes model families correctly", () => { + const cases = [ + { + model: "fun-asr", + expected: { + kind: "async-filetrans", + useSync: false, + path: "/api/v1/services/audio/asr/transcription", + asyncInputStyle: "file_urls", + }, + }, + { + model: "qwen3-asr-flash-filetrans-2025-11-17", + expected: { + kind: "async-filetrans", + useSync: false, + path: "/api/v1/services/audio/asr/transcription", + asyncInputStyle: "file_url", + asyncLanguageStyle: "language", + }, + }, + { + model: "qwen-audio-3.0-asr-flash-filetrans", + expected: { + kind: "async-filetrans", + useSync: false, + asyncInputStyle: "file_urls", + asyncLanguageStyle: "language_hints", + }, + }, + { + model: "qwen3-asr-flash-us", + expected: { + kind: "sync-flash", + useSync: true, + flashFamily: "qwen3", + path: "/api/v1/services/aigc/multimodal-generation/generation", + }, + }, + { + model: "qwen-audio-3.0-asr-flash", + expected: { + kind: "sync-flash", + useSync: true, + flashFamily: "input-audio", + }, + }, + { + model: "qwen3-asr-flash-realtime", + expected: { + kind: "unsupported", + }, + }, + { + model: "foo-asr-flash", + expected: { + kind: "async-filetrans", + useSync: false, + path: "/api/v1/services/audio/asr/transcription", + asyncInputStyle: "file_urls", + }, + }, + ] as const; + + for (const { model, expected } of cases) { + const route = resolveAsrApi(model); + expect(route, model).toMatchObject(expected); + if (expected.kind === "unsupported") { + expect(route.unsupportedReason, model).toMatch(/realtime|streaming|WebSocket/i); + } + } +}); + +test("unknown models default to async-filetrans for backward compatibility", () => { + expect(resolveAsrApi("custom-asr-model")).toMatchObject({ + kind: "async-filetrans", + useSync: false, + }); +}); + +test("inferAudioFormatHint reads extension from url", () => { + expect(inferAudioFormatHint("https://example.com/a.mp3")).toBe("mp3"); + expect(inferAudioFormatHint("oss://bucket/path/file.WAV")).toBe("wav"); + expect(inferAudioFormatHint("https://example.com/a.mpeg?x=1")).toBe("mp3"); + expect(inferAudioFormatHint("https://example.com/noext")).toBe("wav"); + expect(inferAudioFormatHint("data:audio/mpeg;base64,AAA")).toBe("mp3"); + expect(inferAudioFormatHint("data:audio/x-wav;base64,AAA")).toBe("wav"); + expect(inferAudioFormatHint("data:audio/ogg;codecs=opus;base64,AAA")).toBe("ogg"); +}); + +test("buildAsrFlashRequest shapes qwen3 and input-audio bodies", () => { + expect( + buildAsrFlashRequest({ + model: "qwen3-asr-flash", + audioUrl: "https://example.com/a.mp3", + language: "en", + flashFamily: "qwen3", + }), + ).toEqual({ + model: "qwen3-asr-flash", + input: { + messages: [{ role: "user", content: [{ audio: "https://example.com/a.mp3" }] }], + }, + parameters: { asr_options: { language: "en" } }, + }); + + expect( + buildAsrFlashRequest({ + model: "qwen-audio-3.0-asr-flash", + audioUrl: "https://example.com/a.wav", + language: "en", + vocabularyId: "vocab-abc", + flashFamily: "input-audio", + }), + ).toEqual({ + model: "qwen-audio-3.0-asr-flash", + input: { + messages: [ + { + role: "user", + content: [{ type: "input_audio", input_audio: { data: "https://example.com/a.wav" } }], + }, + ], + }, + parameters: { + format: "wav", + sample_rate: "16000", + language_hints: ["en"], + vocabulary_id: "vocab-abc", + }, + }); +}); + +test("buildAsyncAsrLanguageFields maps language by async style", () => { + expect(buildAsyncAsrLanguageFields("language_hints", "zh")).toEqual({ + language_hints: ["zh"], + }); + expect(buildAsyncAsrLanguageFields("language", "zh")).toEqual({ language: "zh" }); + expect(buildAsyncAsrLanguageFields("language", undefined)).toEqual({}); +}); + +test("extractAsrFlashText reads qwen3 choices and input-audio text fields", () => { + expect( + extractAsrFlashText( + { + output: { + choices: [{ message: { content: [{ text: "你好" }] } }], + }, + }, + "qwen3", + ), + ).toBe("你好"); + + expect( + extractAsrFlashText( + { + output: { + text: "Hello World", + output: { sentence: { text: "ignored when text present" } }, + }, + }, + "input-audio", + ), + ).toBe("Hello World"); + + expect( + extractAsrFlashText( + { + output: { + sentence: { text: "top-level sentence" }, + }, + }, + "input-audio", + ), + ).toBe("top-level sentence"); + + expect( + extractAsrFlashText( + { + output: { + output: { sentence: { text: "nested sentence" } }, + }, + }, + "input-audio", + ), + ).toBe("nested sentence"); +}); + +test("collectAsrTranscriptionItems prefers results[] then singular result", () => { + expect( + collectAsrTranscriptionItems({ + results: [{ transcription_url: "https://example.com/a.json", file_url: "https://a.wav" }], + }), + ).toEqual([{ transcription_url: "https://example.com/a.json", file_url: "https://a.wav" }]); + + expect( + collectAsrTranscriptionItems({ + result: { transcription_url: "https://example.com/qwen3.json" }, + }), + ).toEqual([{ transcription_url: "https://example.com/qwen3.json", subtask_status: "SUCCEEDED" }]); + + expect(collectAsrTranscriptionItems({})).toEqual([]); +}); diff --git a/packages/runtime/src/pipeline/steps/bl-api.ts b/packages/runtime/src/pipeline/steps/bl-api.ts index 886ad5b..8206068 100644 --- a/packages/runtime/src/pipeline/steps/bl-api.ts +++ b/packages/runtime/src/pipeline/steps/bl-api.ts @@ -10,6 +10,11 @@ import { taskPath, speechSynthesizePath, speechRecognizePath, + resolveAsrApi, + buildAsrFlashRequest, + buildAsyncAsrLanguageFields, + collectAsrTranscriptionItems, + extractAsrFlashText, stripUndefined, resolveBooleanFlag, resolveWatermark, @@ -23,6 +28,7 @@ import { type DashScopeTTSRequest, type DashScopeTTSResponse, type DashScopeASRRequest, + type DashScopeASRTaskResult, type ChatMessageContent, isLocalFile, } from "bailian-cli-core"; @@ -573,27 +579,103 @@ export async function speechRecognize( }); } + const model = input.model || "fun-asr"; + const route = resolveAsrApi(model); + if (route.kind === "unsupported") { + throw new PipelineError( + "invalid_input", + route.unsupportedReason ?? `Unsupported ASR model: ${model}`, + { + step: "speech/recognize", + }, + ); + } + + if (route.kind === "sync-flash") { + if (rawUrls.length !== 1) { + throw new PipelineError( + "invalid_input", + `Model "${model}" is a sync Flash ASR model and accepts exactly one url (got ${rawUrls.length})`, + { step: "speech/recognize" }, + ); + } + const unsupportedFlags: string[] = []; + if (input.diarization) unsupportedFlags.push("diarization"); + if (input["speaker-count"] !== undefined) unsupportedFlags.push("speaker-count"); + // input-audio Flash supports vocabulary_id; qwen3 sync Flash does not + if (route.flashFamily === "qwen3" && input["vocabulary-id"] !== undefined) { + unsupportedFlags.push("vocabulary-id"); + } + if (input["channel-id"] !== undefined) unsupportedFlags.push("channel-id"); + if (unsupportedFlags.length > 0) { + throw new PipelineError( + "invalid_input", + `Model "${model}" uses sync Flash ASR and does not support: ${unsupportedFlags.join(", ")}`, + { step: "speech/recognize" }, + ); + } + } + if ( + route.kind === "async-filetrans" && + route.asyncInputStyle === "file_url" && + rawUrls.length !== 1 + ) { + throw new PipelineError( + "invalid_input", + `Model "${model}" accepts exactly one url (got ${rawUrls.length})`, + { step: "speech/recognize" }, + ); + } + // Resolve local files to upload URLs const fileUrls: string[] = []; - for (const u of rawUrls) { - if (isLocalFile(u)) { + for (const audioUrl of rawUrls) { + if (isLocalFile(audioUrl)) { fileUrls.push( - await env.client.uploadFile(u, input.model || "fun-asr", { + await env.client.uploadFile(audioUrl, model, { signal: ctx.signal, }), ); } else { - fileUrls.push(u); + fileUrls.push(audioUrl); } } - const model = input.model || "fun-asr"; + if (route.kind === "sync-flash") { + const flashFamily = route.flashFamily!; + const body = buildAsrFlashRequest({ + model, + audioUrl: fileUrls[0]!, + language: input.language, + vocabularyId: input["vocabulary-id"], + flashFamily, + }); + const response = await env.client.requestJson>({ + path: route.path, + method: "POST", + headers: { "X-DashScope-SSE": "disable" }, + body, + signal: ctx.signal, + }); + return { + text: extractAsrFlashText(response, flashFamily), + model, + mode: "sync", + raw: response, + }; + } + + const languageFields = buildAsyncAsrLanguageFields( + route.asyncLanguageStyle ?? "language_hints", + input.language, + ); const body: DashScopeASRRequest = { model, - input: { file_urls: fileUrls }, + input: + route.asyncInputStyle === "file_url" ? { file_url: fileUrls[0]! } : { file_urls: fileUrls }, parameters: { channel_id: input["channel-id"] !== undefined ? [input["channel-id"]] : undefined, - language_hints: input.language ? [input.language] : undefined, + ...languageFields, diarization_enabled: input.diarization, speaker_count: input["speaker-count"], vocabulary_id: input["vocabulary-id"], @@ -601,9 +683,8 @@ export async function speechRecognize( }; stripUndefined(body.parameters as Record); - const url = speechRecognizePath(); const asyncResp = await env.client.requestJson({ - path: url, + path: speechRecognizePath(), method: "POST", body, async: true, @@ -614,7 +695,65 @@ export async function speechRecognize( const pollIntervalMs = (input["poll-interval"] ?? 2) * 1000; const timeoutMs = (ctx.timeoutSeconds ?? 300) * 1000; - return await pollTaskWithOptions(env, taskId, pollIntervalMs, timeoutMs, ctx); + // ASR polling reads original output, avoids generic flatten (avoids transcription_url polluting media urls) + const asrTask = await pollAsrTaskWithOptions(env, taskId, pollIntervalMs, timeoutMs, ctx); + const transcriptionItems = collectAsrTranscriptionItems(asrTask.output); + + const base: Record = { + task_id: asrTask.output.task_id, + task_status: asrTask.output.task_status, + request_id: asrTask.request_id, + mode: "async", + model, + }; + if (asrTask.output.results) base.results = asrTask.output.results; + if (asrTask.output.result) { + base.result = asrTask.output.result; + if (typeof asrTask.output.result.transcription_url === "string") { + base.transcription_url = asrTask.output.result.transcription_url; + } + } + if (asrTask.output.task_metrics) base.task_metrics = asrTask.output.task_metrics; + if (asrTask.usage) base.usage = asrTask.usage; + + if (transcriptionItems.length === 0) { + return base; + } + + const texts: string[] = []; + const transcripts: Record[] = []; + for (const item of transcriptionItems) { + if (!item.transcription_url) continue; + const transRes = await fetch(item.transcription_url, { signal: ctx.signal }); + if (!transRes.ok) { + throw new PipelineError( + "async_task_failed", + `Failed to download transcription: HTTP ${transRes.status}`, + { step: "speech/recognize", details: { taskId, url: item.transcription_url } }, + ); + } + const transData = (await transRes.json()) as Record; + transcripts.push(transData); + const transcriptList = transData.transcripts as + | Array<{ text?: string; sentences?: Array<{ text?: string }> }> + | undefined; + if (!transcriptList?.length) continue; + for (const transcript of transcriptList) { + if (transcript.sentences?.length) { + for (const sentence of transcript.sentences) { + if (sentence.text) texts.push(sentence.text); + } + } else if (transcript.text) { + texts.push(transcript.text); + } + } + } + + return { + ...base, + text: texts.join("\n"), + transcripts, + }; } // --- Shared: task polling --- @@ -640,7 +779,7 @@ function flattenTaskResponse(resp: DashScopeTaskResponse): Record 0) flat.urls = urls; } if (output.results) { - const urls = output.results.map((r) => r.url).filter(Boolean); + const urls = output.results.map((item) => item.url).filter(Boolean); if (urls.length > 0 && !flat.urls) flat.urls = urls; } if (output.task_metrics) flat.task_metrics = output.task_metrics; @@ -658,13 +797,13 @@ async function pollTask( return await pollTaskWithOptions(env, taskId, pollIntervalMs, timeoutMs, ctx); } -async function pollTaskWithOptions( +async function pollUntilSucceeded( env: PipelineEnv, taskId: string, pollIntervalMs: number, timeoutMs: number, ctx?: StepContext, -): Promise> { +): Promise { const started = Date.now(); let attempt = 0; @@ -687,7 +826,7 @@ async function pollTaskWithOptions( const status = result.output.task_status; if (status === "SUCCEEDED") { - return flattenTaskResponse(result); + return result; } if (status === "FAILED") { @@ -718,6 +857,33 @@ async function pollTaskWithOptions( } } +async function pollTaskWithOptions( + env: PipelineEnv, + taskId: string, + pollIntervalMs: number, + timeoutMs: number, + ctx?: StepContext, +): Promise> { + return flattenTaskResponse(await pollUntilSucceeded(env, taskId, pollIntervalMs, timeoutMs, ctx)); +} + +/** ASR task polling: preserve original output (includes results[] / result.transcription_url). */ +async function pollAsrTaskWithOptions( + env: PipelineEnv, + taskId: string, + pollIntervalMs: number, + timeoutMs: number, + ctx?: StepContext, +): Promise { + return (await pollUntilSucceeded( + env, + taskId, + pollIntervalMs, + timeoutMs, + ctx, + )) as DashScopeASRTaskResult; +} + function delay(ms: number, signal?: AbortSignal): Promise { if (!signal) return new Promise((resolve) => setTimeout(resolve, ms)); return new Promise((resolve, reject) => { diff --git a/packages/runtime/tests/speech-recognize-pipeline.test.ts b/packages/runtime/tests/speech-recognize-pipeline.test.ts new file mode 100644 index 0000000..3ebef52 --- /dev/null +++ b/packages/runtime/tests/speech-recognize-pipeline.test.ts @@ -0,0 +1,184 @@ +import { expect, test } from "vite-plus/test"; +import type { Client } from "bailian-cli-core"; +import { PipelineError } from "../src/pipeline/errors.ts"; +import type { PipelineEnv } from "../src/pipeline/bl-config.ts"; +import { speechRecognize } from "../src/pipeline/steps/bl-api.ts"; +import type { StepContext } from "../src/pipeline/types.ts"; + +type CapturedRequest = { + path?: string; + method?: string; + headers?: Record; + body?: Record; + async?: boolean; +}; + +function makeEnv(requestJsonImpl?: (opts: CapturedRequest) => Promise): { + env: PipelineEnv; + captured: CapturedRequest[]; +} { + const captured: CapturedRequest[] = []; + const client = { + uploadFile: async (source: string) => source, + requestJson: async (opts: CapturedRequest) => { + captured.push(opts); + if (requestJsonImpl) return requestJsonImpl(opts); + return { output: { text: "ok" } }; + }, + } as unknown as Client; + + return { + env: { + client, + settings: { quiet: true, output: "json" } as PipelineEnv["settings"], + }, + captured, + }; +} + +function makeCtx(): StepContext { + return { dryRun: false, signal: new AbortController().signal }; +} + +test("pipeline speechRecognize routes input-audio flash to sync multimodal endpoint", async () => { + const { env, captured } = makeEnv(); + const result = (await speechRecognize( + env, + { + url: "https://example.com/a.wav", + model: "qwen-audio-3.0-asr-flash", + language: "en", + "vocabulary-id": "vocab-1", + }, + makeCtx(), + )) as { mode?: string; text?: string }; + + expect(result.mode).toBe("sync"); + expect(result.text).toBe("ok"); + expect(captured).toHaveLength(1); + expect(captured[0]?.path).toBe("/api/v1/services/aigc/multimodal-generation/generation"); + expect(captured[0]?.headers?.["X-DashScope-SSE"]).toBe("disable"); + expect(captured[0]?.body).toMatchObject({ + model: "qwen-audio-3.0-asr-flash", + parameters: { + format: "wav", + language_hints: ["en"], + vocabulary_id: "vocab-1", + }, + }); +}); + +test("pipeline speechRecognize maps qwen3-filetrans language to parameters.language", async () => { + const { env, captured } = makeEnv(async (opts) => { + if (opts.async || opts.method === "POST") { + return { output: { task_id: "task-1", task_status: "PENDING" } }; + } + return { + output: { task_id: "task-1", task_status: "SUCCEEDED", results: [] }, + request_id: "r1", + }; + }); + + await speechRecognize( + env, + { + url: "https://example.com/a.wav", + model: "qwen3-asr-flash-filetrans", + language: "zh", + "poll-interval": 0, + }, + makeCtx(), + ); + + expect(captured[0]?.path).toBe("/api/v1/services/audio/asr/transcription"); + expect(captured[0]?.async).toBe(true); + expect(captured[0]?.body).toMatchObject({ + model: "qwen3-asr-flash-filetrans", + input: { file_url: "https://example.com/a.wav" }, + parameters: { language: "zh" }, + }); + expect( + (captured[0]?.body?.parameters as Record | undefined)?.language_hints, + ).toBeUndefined(); +}); + +test("pipeline speechRecognize rejects realtime models before requesting", async () => { + const { env, captured } = makeEnv(); + await expect( + speechRecognize( + env, + { url: "https://example.com/a.wav", model: "qwen3-asr-flash-realtime" }, + makeCtx(), + ), + ).rejects.toBeInstanceOf(PipelineError); + expect(captured).toHaveLength(0); +}); + +test("pipeline speechRecognize rejects multiple urls for sync flash", async () => { + const { env, captured } = makeEnv(); + await expect( + speechRecognize( + env, + { + url: ["https://example.com/a.wav", "https://example.com/b.wav"], + model: "fun-asr-flash-2026-06-15", + }, + makeCtx(), + ), + ).rejects.toBeInstanceOf(PipelineError); + expect(captured).toHaveLength(0); +}); + +test("pipeline speechRecognize downloads qwen3 singular result.transcription_url", async () => { + const originalFetch = globalThis.fetch; + const transcriptionUrl = "https://example.com/transcription.json"; + globalThis.fetch = (async (input: RequestInfo | URL) => { + const url = typeof input === "string" ? input : input instanceof URL ? input.href : input.url; + expect(url).toBe(transcriptionUrl); + return new Response( + JSON.stringify({ + transcripts: [{ text: "pipeline hello", sentences: [{ text: "pipeline hello" }] }], + }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ); + }) as typeof fetch; + + try { + const { env, captured } = makeEnv(async (opts) => { + if (opts.async || opts.method === "POST") { + return { output: { task_id: "task-1", task_status: "PENDING" } }; + } + return { + output: { + task_id: "task-1", + task_status: "SUCCEEDED", + result: { transcription_url: transcriptionUrl }, + }, + request_id: "r1", + }; + }); + + const result = (await speechRecognize( + env, + { + url: "https://example.com/a.wav", + model: "qwen3-asr-flash-filetrans", + "poll-interval": 0, + }, + makeCtx(), + )) as { + mode?: string; + text?: string; + transcription_url?: string; + result?: { transcription_url?: string }; + }; + + expect(captured[0]?.async).toBe(true); + expect(result.mode).toBe("async"); + expect(result.text).toBe("pipeline hello"); + expect(result.transcription_url).toBe(transcriptionUrl); + expect(result.result?.transcription_url).toBe(transcriptionUrl); + } finally { + globalThis.fetch = originalFetch; + } +}); diff --git a/skills/bailian-cli/SKILL.md b/skills/bailian-cli/SKILL.md index 946d02a..c496a43 100644 --- a/skills/bailian-cli/SKILL.md +++ b/skills/bailian-cli/SKILL.md @@ -71,6 +71,7 @@ Use this table only after the decision table in [`bailian-protocol`](../bailian- | Bailian pipeline workflow (a step in a bl flow) | `bl pipeline run` / `validate` | JSON/YAML workflow definitions | | Bailian rate limits / quota | `bl quota list` / `check` / `request` | Console auth; class 2 — ask which product first if unnamed | | Bailian free tier / usage stats | `bl usage free` / `stats` / `freetier` | Console auth; class 2 — ask which product first if unnamed | +| Bailian Token Plan quota usage | `bl usage token-plan` | Console auth; class 2 — ask which product first if unnamed | | Console API (advanced) | `bl console call` | Console auth | | Bailian workspace listing | `bl workspace list` | Console auth | | Image / video / speech / omni / vision | → skill `bailian-gen` | Fallback: `bl image\|video\|speech\|omni\|vision --help` | diff --git a/skills/bailian-cli/reference/index.md b/skills/bailian-cli/reference/index.md index 7d748bb..2fa87ae 100644 --- a/skills/bailian-cli/reference/index.md +++ b/skills/bailian-cli/reference/index.md @@ -9,65 +9,66 @@ Use this index for the skill-scoped quick index and global flags. ## Quick index -| Command | Authentication | Description | Detail | -| ------------------------------- | -------------- | ---------------------------------------------------------------------------------------------- | ------------------------------ | -| `bl advisor recommend` | API Key | Recommend the best models for your use case (intent analysis → candidate recall → LLM ranking) | [advisor.md](advisor.md) | -| `bl app call` | API Key | Call a Bailian application (agent or workflow) | [app.md](app.md) | -| `bl app list` | Console | List Bailian applications | [app.md](app.md) | -| `bl auth generate-access-token` | No Auth | Generate a CLI access token using OpenAPI AK/SK | [auth.md](auth.md) | -| `bl auth login` | No Auth | Authenticate with API key, console browser login, or OpenAPI AK/SK (credentials can coexist) | [auth.md](auth.md) | -| `bl auth logout` | No Auth | Clear stored credentials; full logout also clears the model Base URL | [auth.md](auth.md) | -| `bl auth status` | No Auth | Show current authentication state | [auth.md](auth.md) | -| `bl config agent` | No Auth | Configure a coding agent to use DashScope API | [config.md](config.md) | -| `bl config list` | No Auth | List config profiles and show the active profile | [config.md](config.md) | -| `bl config set` | No Auth | Set a config value | [config.md](config.md) | -| `bl config show` | No Auth | Display current configuration | [config.md](config.md) | -| `bl config ui` | No Auth | Open a local web UI to manage config profiles | [config.md](config.md) | -| `bl config use` | No Auth | Set the active config profile | [config.md](config.md) | -| `bl console call` | Console | Call a Bailian console API via the CLI gateway | [console.md](console.md) | -| `bl file upload` | API Key | Upload a local file to DashScope temporary storage (48h) | [file.md](file.md) | -| `bl knowledge chat` | API Key | Chat with a Bailian knowledge base (RAG Q&A with streaming) | [knowledge.md](knowledge.md) | -| `bl knowledge retrieve` | API Key | Retrieve from a Bailian knowledge base (deprecated, use `search` instead) | [knowledge.md](knowledge.md) | -| `bl knowledge search` | API Key | Search a Bailian knowledge base (RAG semantic retrieval) | [knowledge.md](knowledge.md) | -| `bl mcp call` | API Key | Call a tool on an MCP server (tools/call) | [mcp.md](mcp.md) | -| `bl mcp list` | Console | List MCP servers activated under your Bailian account | [mcp.md](mcp.md) | -| `bl mcp tools` | API Key | List tools exposed by an MCP server (tools/list) | [mcp.md](mcp.md) | -| `bl memory add` | API Key | Add memory from messages or custom content | [memory.md](memory.md) | -| `bl memory delete` | API Key | Delete a memory node | [memory.md](memory.md) | -| `bl memory list` | API Key | List memory nodes for a user | [memory.md](memory.md) | -| `bl memory profile create` | API Key | Create a user profile schema for memory profiling | [memory.md](memory.md) | -| `bl memory profile get` | API Key | Get user profile by schema ID and user ID | [memory.md](memory.md) | -| `bl memory search` | API Key | Search memory nodes by query or messages | [memory.md](memory.md) | -| `bl memory update` | API Key | Update a memory node content | [memory.md](memory.md) | -| `bl model list` | Console | Browse model families or show detailed model info in the Bailian model marketplace | [model.md](model.md) | -| `bl pipeline run` | No Auth | Run a pipeline workflow definition | [pipeline.md](pipeline.md) | -| `bl pipeline validate` | No Auth | Validate a pipeline definition without executing | [pipeline.md](pipeline.md) | -| `bl plugin install` | No Auth | Install or upgrade an allowlisted Command Pack | [plugin.md](plugin.md) | -| `bl plugin link` | No Auth | Link an allowlisted local Command Pack for development | [plugin.md](plugin.md) | -| `bl plugin list` | No Auth | List installed Command Packs and their load status | [plugin.md](plugin.md) | -| `bl plugin remove` | No Auth | Remove an installed Command Pack | [plugin.md](plugin.md) | -| `bl quota check` | Console | Check current usage against rate limits | [quota.md](quota.md) | -| `bl quota history` | Console | View quota change history | [quota.md](quota.md) | -| `bl quota list` | Console | View model RPM/TPM rate limits | [quota.md](quota.md) | -| `bl quota request` | Console | Request a temporary quota increase | [quota.md](quota.md) | -| `bl search web` | API Key | Search the web using DashScope MCP WebSearch service | [search.md](search.md) | -| `bl skill add` | No Auth | Install skills from the Bailian skill registry into local agents | [skill.md](skill.md) | -| `bl skill init` | No Auth | Install all bailian-\* skills (one-shot bootstrap for new environments) | [skill.md](skill.md) | -| `bl skill list` | No Auth | List registry skills and diff against local installs | [skill.md](skill.md) | -| `bl skill remove` | No Auth | Remove locally installed skills (registry is untouched) | [skill.md](skill.md) | -| `bl skill update` | No Auth | Update installed skills to the latest registry versions | [skill.md](skill.md) | -| `bl text chat` | API Key | Send a chat completion (OpenAI compatible, DashScope) | [text.md](text.md) | -| `bl token-plan add-member` | AK/SK | Add a member to a Token Plan organization | [token-plan.md](token-plan.md) | -| `bl token-plan assign-seats` | AK/SK | Batch assign Token Plan seats to members | [token-plan.md](token-plan.md) | -| `bl token-plan create-key` | AK/SK | Create a Token Plan API key for a seat | [token-plan.md](token-plan.md) | -| `bl token-plan list-seats` | AK/SK | List Token Plan subscription seat details | [token-plan.md](token-plan.md) | -| `bl update` | No Auth | Update the CLI to the latest or a specified version | [update.md](update.md) | -| `bl usage free` | Console | Query free-tier quota for models (all models if --model is omitted) | [usage.md](usage.md) | -| `bl usage freetier` | Console | Enable or disable auto-stop for free-tier models. Enables by default; use --off to disable | [usage.md](usage.md) | -| `bl usage stats` | Console | Query model usage statistics | [usage.md](usage.md) | -| `bl usage summary` | Console | Show a unified usage summary: free-tier quota and recent usage overview | [usage.md](usage.md) | -| `bl workspace init` | No Auth | Initialize Bailian workspace and activate postpaid services | [workspace.md](workspace.md) | -| `bl workspace list` | Console | List all workspaces | [workspace.md](workspace.md) | +| Command | Description | Detail | +| ------------------------------- | ---------------------------------------------------------------------------------------------- | ------------------------------ | +| `bl advisor recommend` | Recommend the best models for your use case (intent analysis → candidate recall → LLM ranking) | [advisor.md](advisor.md) | +| `bl app call` | Call a Bailian application (agent or workflow) | [app.md](app.md) | +| `bl app list` | List Bailian applications | [app.md](app.md) | +| `bl auth generate-access-token` | Generate a CLI access token using OpenAPI AK/SK | [auth.md](auth.md) | +| `bl auth login` | Authenticate with API key, console browser login, or OpenAPI AK/SK (credentials can coexist) | [auth.md](auth.md) | +| `bl auth logout` | Clear stored credentials; full logout also clears the model Base URL | [auth.md](auth.md) | +| `bl auth status` | Show current authentication state | [auth.md](auth.md) | +| `bl config agent` | Configure a coding agent to use DashScope API | [config.md](config.md) | +| `bl config list` | List config profiles and show the active profile | [config.md](config.md) | +| `bl config set` | Set a config value | [config.md](config.md) | +| `bl config show` | Display current configuration | [config.md](config.md) | +| `bl config ui` | Open a local web UI to manage config profiles | [config.md](config.md) | +| `bl config use` | Set the active config profile | [config.md](config.md) | +| `bl console call` | Call a Bailian console API via the CLI gateway | [console.md](console.md) | +| `bl file upload` | Upload a local file to DashScope temporary storage (48h) | [file.md](file.md) | +| `bl knowledge chat` | Chat with a Bailian knowledge base (RAG Q&A with streaming) | [knowledge.md](knowledge.md) | +| `bl knowledge retrieve` | Retrieve from a Bailian knowledge base (deprecated, use `search` instead) | [knowledge.md](knowledge.md) | +| `bl knowledge search` | Search a Bailian knowledge base (RAG semantic retrieval) | [knowledge.md](knowledge.md) | +| `bl mcp call` | Call a tool on an MCP server (tools/call) | [mcp.md](mcp.md) | +| `bl mcp list` | List MCP servers activated under your Bailian account | [mcp.md](mcp.md) | +| `bl mcp tools` | List tools exposed by an MCP server (tools/list) | [mcp.md](mcp.md) | +| `bl memory add` | Add memory from messages or custom content | [memory.md](memory.md) | +| `bl memory delete` | Delete a memory node | [memory.md](memory.md) | +| `bl memory list` | List memory nodes for a user | [memory.md](memory.md) | +| `bl memory profile create` | Create a user profile schema for memory profiling | [memory.md](memory.md) | +| `bl memory profile get` | Get user profile by schema ID and user ID | [memory.md](memory.md) | +| `bl memory search` | Search memory nodes by query or messages | [memory.md](memory.md) | +| `bl memory update` | Update a memory node content | [memory.md](memory.md) | +| `bl model list` | Browse model families or show detailed model info in the Bailian model marketplace | [model.md](model.md) | +| `bl pipeline run` | Run a pipeline workflow definition | [pipeline.md](pipeline.md) | +| `bl pipeline validate` | Validate a pipeline definition without executing | [pipeline.md](pipeline.md) | +| `bl plugin install` | Install or upgrade an allowlisted Command Pack | [plugin.md](plugin.md) | +| `bl plugin link` | Link an allowlisted local Command Pack for development | [plugin.md](plugin.md) | +| `bl plugin list` | List installed Command Packs and their load status | [plugin.md](plugin.md) | +| `bl plugin remove` | Remove an installed Command Pack | [plugin.md](plugin.md) | +| `bl quota check` | Check current usage against rate limits | [quota.md](quota.md) | +| `bl quota history` | View quota change history | [quota.md](quota.md) | +| `bl quota list` | View model RPM/TPM rate limits | [quota.md](quota.md) | +| `bl quota request` | Request a temporary quota increase | [quota.md](quota.md) | +| `bl search web` | Search the web using DashScope MCP WebSearch service | [search.md](search.md) | +| `bl skill add` | Install skills from the Bailian skill registry into local agents | [skill.md](skill.md) | +| `bl skill init` | Install all bailian-\* skills (one-shot bootstrap for new environments) | [skill.md](skill.md) | +| `bl skill list` | List registry skills and diff against local installs | [skill.md](skill.md) | +| `bl skill remove` | Remove locally installed skills (registry is untouched) | [skill.md](skill.md) | +| `bl skill update` | Update installed skills to the latest registry versions | [skill.md](skill.md) | +| `bl text chat` | Send a chat completion (OpenAI compatible, DashScope) | [text.md](text.md) | +| `bl token-plan add-member` | Add a member to a Token Plan organization | [token-plan.md](token-plan.md) | +| `bl token-plan assign-seats` | Batch assign Token Plan seats to members | [token-plan.md](token-plan.md) | +| `bl token-plan create-key` | Create a Token Plan API key for a seat | [token-plan.md](token-plan.md) | +| `bl token-plan list-seats` | List Token Plan subscription seat details | [token-plan.md](token-plan.md) | +| `bl update` | Update the CLI to the latest or a specified version | [update.md](update.md) | +| `bl usage free` | Query free-tier quota for models (all models if --model is omitted) | [usage.md](usage.md) | +| `bl usage freetier` | Enable or disable auto-stop for free-tier models. Enables by default; use --off to disable | [usage.md](usage.md) | +| `bl usage stats` | Query model usage statistics | [usage.md](usage.md) | +| `bl usage summary` | Show a unified usage summary: free-tier quota and recent usage overview | [usage.md](usage.md) | +| `bl usage token-plan` | Show Token Plan quota usage | [usage.md](usage.md) | +| `bl workspace init` | Initialize Bailian workspace and activate postpaid services | [workspace.md](workspace.md) | +| `bl workspace list` | List all workspaces | [workspace.md](workspace.md) | ## By group @@ -91,7 +92,7 @@ Use this index for the skill-scoped quick index and global flags. | `text` | `chat` | [text.md](text.md) | | `token-plan` | `add-member`, `assign-seats`, `create-key`, `list-seats` | [token-plan.md](token-plan.md) | | `update` | `(root)` | [update.md](update.md) | -| `usage` | `free`, `freetier`, `stats`, `summary` | [usage.md](usage.md) | +| `usage` | `free`, `freetier`, `stats`, `summary`, `token-plan` | [usage.md](usage.md) | | `workspace` | `init`, `list` | [workspace.md](workspace.md) | ## Global flags diff --git a/skills/bailian-cli/reference/usage.md b/skills/bailian-cli/reference/usage.md index d8087a5..1fed601 100644 --- a/skills/bailian-cli/reference/usage.md +++ b/skills/bailian-cli/reference/usage.md @@ -7,12 +7,13 @@ Index: [index.md](index.md) ## Commands in this group -| Command | Authentication | Description | -| ------------------- | -------------- | ------------------------------------------------------------------------------------------ | -| `bl usage free` | Console | Query free-tier quota for models (all models if --model is omitted) | -| `bl usage freetier` | Console | Enable or disable auto-stop for free-tier models. Enables by default; use --off to disable | -| `bl usage stats` | Console | Query model usage statistics | -| `bl usage summary` | Console | Show a unified usage summary: free-tier quota and recent usage overview | +| Command | Description | +| --------------------- | ------------------------------------------------------------------------------------------ | +| `bl usage free` | Query free-tier quota for models (all models if --model is omitted) | +| `bl usage freetier` | Enable or disable auto-stop for free-tier models. Enables by default; use --off to disable | +| `bl usage stats` | Query model usage statistics | +| `bl usage summary` | Show a unified usage summary: free-tier quota and recent usage overview | +| `bl usage token-plan` | Show Token Plan quota usage | ## Command details @@ -203,3 +204,30 @@ bl usage summary --days 30 ```bash bl usage summary --output json ``` + +### `bl usage token-plan` + +| Field | Value | +| --------------- | ----------------------------- | +| **Name** | `usage token-plan` | +| **Description** | Show Token Plan quota usage | +| **Usage** | `bl usage token-plan [flags]` | + +#### Flags + +| Flag | Type | Required | Description | +| ------------------------------ | ------ | -------- | -------------------------------------------------------- | +| `--console-region ` | string | no | Console gateway region (e.g. cn-beijing, ap-southeast-1) | +| `--console-site ` | string | no | Console site: domestic, international | +| `--console-switch-agent ` | number | no | Switch agent UID for delegated access | +| `--workspace-id ` | string | no | Workspace ID (env: BAILIAN_WORKSPACE_ID) | + +#### Examples + +```bash +bl usage token-plan +``` + +```bash +bl usage token-plan --output json +``` diff --git a/skills/bailian-gen/SKILL.md b/skills/bailian-gen/SKILL.md index 14297c2..0ed90b2 100644 --- a/skills/bailian-gen/SKILL.md +++ b/skills/bailian-gen/SKILL.md @@ -39,6 +39,8 @@ description: >- | A/V understanding (files the host can't play) | `bl omni --video` / `--audio` | `qwen3.5-omni-plus` | | Image/video describe (user names Bailian) | `bl vision describe` | `qwen-vl-max`; host-first for plain image Q&A | +For ASR model selection, keep `fun-asr` (or other `*-filetrans`) for long recordings, repeated files, speaker diarization, or asynchronous task IDs. For one local or remote audio file up to about five minutes when the user asks for low-latency Flash models, use `--model fun-asr-flash-2026-06-15`, `--model qwen-audio-3.0-asr-flash`, or `--model qwen3-asr-flash`. Flash recognition is synchronous and accepts exactly one file per call. + Flags, usage, and examples: see [`reference/`](reference/index.md) or `bl --help` — do not guess flags. ## Local files (mandatory) diff --git a/skills/bailian-gen/reference/index.md b/skills/bailian-gen/reference/index.md index e1c72ef..85cab64 100644 --- a/skills/bailian-gen/reference/index.md +++ b/skills/bailian-gen/reference/index.md @@ -14,7 +14,7 @@ Use this index for the skill-scoped quick index and global flags. | `bl image edit` | API Key | Edit an existing image with text instructions (Qwen-Image / Wan 2.7) | [image.md](image.md) | | `bl image generate` | API Key | Generate images (Qwen-Image / wan2.x) | [image.md](image.md) | | `bl omni` | API Key | Multimodal chat with text + audio output (Qwen-Omni) | [omni.md](omni.md) | -| `bl speech recognize` | API Key | Recognize speech from audio files (FunAudio-ASR) | [speech.md](speech.md) | +| `bl speech recognize` | API Key | Recognize speech from audio files (FunAudio-ASR / Qwen-ASR Flash) | [speech.md](speech.md) | | `bl speech synthesize` | API Key | Synthesize speech from text (CosyVoice TTS) | [speech.md](speech.md) | | `bl video download` | API Key | Download a completed video by task ID | [video.md](video.md) | | `bl video edit` | API Key | Edit a video with happyhorse-1.0-video-edit (style transfer, object replacement, etc.) | [video.md](video.md) | diff --git a/skills/bailian-gen/reference/speech.md b/skills/bailian-gen/reference/speech.md index 080ce6b..ba448a9 100644 --- a/skills/bailian-gen/reference/speech.md +++ b/skills/bailian-gen/reference/speech.md @@ -7,38 +7,38 @@ Index: [index.md](index.md) ## Commands in this group -| Command | Authentication | Description | -| ---------------------- | -------------- | ------------------------------------------------ | -| `bl speech recognize` | API Key | Recognize speech from audio files (FunAudio-ASR) | -| `bl speech synthesize` | API Key | Synthesize speech from text (CosyVoice TTS) | +| Command | Authentication | Description | +| ---------------------- | -------------- | ----------------------------------------------------------------- | +| `bl speech recognize` | API Key | Recognize speech from audio files (FunAudio-ASR / Qwen-ASR Flash) | +| `bl speech synthesize` | API Key | Synthesize speech from text (CosyVoice TTS) | ## Command details ### `bl speech recognize` -| Field | Value | -| ------------------ | ------------------------------------------------ | -| **Name** | `speech recognize` | -| **Description** | Recognize speech from audio files (FunAudio-ASR) | -| **Authentication** | API Key | -| **Usage** | `bl speech recognize --url [flags]` | +| Field | Value | +| ------------------ | ----------------------------------------------------------------- | +| **Name** | `speech recognize` | +| **Description** | Recognize speech from audio files (FunAudio-ASR / Qwen-ASR Flash) | +| **Authentication** | API Key | +| **Usage** | `bl speech recognize --url [flags]` | #### Flags -| Flag | Type | Required | Description | -| --------------------------- | ------ | -------- | ------------------------------------------------------- | -| `--url ` | array | yes | Audio file URL or local file path (repeatable, max 100) | -| `--model ` | string | no | Model ID (default: fun-asr) | -| `--language ` | string | no | Language hint (e.g. zh, en, ja) | -| `--diarization` | switch | no | Enable automatic speaker diarization | -| `--speaker-count ` | number | no | Expected number of speakers (requires --diarization) | -| `--vocabulary-id ` | string | no | Hot-word vocabulary ID for improved accuracy | -| `--channel-id ` | number | no | Audio channel ID (default: 0) | -| `--out ` | string | no | Save full transcription result to JSON file | -| `--async` | switch | no | Return async task id without waiting | -| `--poll-interval ` | number | no | Polling interval in seconds (default: 2) | -| `--api-key ` | string | no | API key | -| `--base-url ` | string | no | API base URL | +| Flag | Type | Required | Description | +| --------------------------- | ------ | -------- | ------------------------------------------------------------------------------------------------------------------------------------------- | +| `--url ` | array | yes | Audio file URL or local file path (repeatable, max 100) | +| `--model ` | string | no | Model ID (default: fun-asr). Async: fun-asr / _-filetrans / paraformer-_; sync: qwen3-asr-flash* / fun-asr-flash* / qwen-audio-\*-asr-flash | +| `--language ` | string | no | Language hint (e.g. zh, en, ja). Classic async/input-audio: language_hints; qwen3-filetrans: language; qwen3 sync: asr_options.language | +| `--diarization` | switch | no | Enable automatic speaker diarization | +| `--speaker-count ` | number | no | Expected number of speakers (requires --diarization) | +| `--vocabulary-id ` | string | no | Hot-word vocabulary ID for improved accuracy | +| `--channel-id ` | number | no | Audio channel ID (default: 0) | +| `--out ` | string | no | Save full transcription result to JSON file | +| `--async` | switch | no | Return async task id without waiting | +| `--poll-interval ` | number | no | Polling interval in seconds (default: 2) | +| `--api-key ` | string | no | API key | +| `--base-url ` | string | no | API base URL | #### Examples @@ -70,6 +70,10 @@ bl speech recognize --url https://example.com/audio.mp3 --out result.json bl speech recognize --url https://example.com/audio.mp3 --async --quiet ``` +```bash +bl speech recognize --url https://example.com/audio.mp3 --model qwen-audio-3.0-asr-flash --language en +``` + ### `bl speech synthesize` | Field | Value |