返回源码地图

packages/experimental/api-speech-to-text/src/index.ts

main snapshot · da00f7f5358f · 正文引用章节 17;完整原文可核对,不声称全文件人工逐行审计

完整原文供逐行核对;页面收录不代表每行都经过人工语义审核。MIT 许可见 许可证。

1/** Authenticated, cancellation-aware Client access to the speech capability. */
2import { Context } from '@deepseek-ai/cordis'
3import z from '@deepseek-ai/schemastery'
4import { Remote, RemoteError, TypertRemoteService } from '@deepseek-ai/dsh-typert-protocol'
5import type {} from '@deepseek-ai/dsh-experimental-speech-to-text'
6import type { SpeechPreparationOptions, SpeechProviderId, SpeechSelectionPatch, Transcript } from '@deepseek-ai/dsh-experimental-speech-to-text/types'
7import type { SpeechCatalog, TranscriptionRequest } from './types.ts'
8import { validateWave } from '@deepseek-ai/dsh-experimental-speech-to-text/wave'
9
10export type * from './types.ts'
11
12declare module '@deepseek-ai/cordis' {
13 interface Context {
14 /** Experimental speech Remote controller. */
15 speechController: SpeechController
16 }
17}
18
19/** Limits applied before decoding or calling a provider. */
20export interface Config {
21 /** Maximum decoded WAV bytes per request. */
22 maxAudioBytes: number
23 /** Maximum PCM recording duration in seconds. */
24 maxDurationSeconds: number
25}
26
27/** Speech calls never activate or submit to an Agent. */
28export default class SpeechController extends TypertRemoteService {
29 static inject = ['speechToText', 'typert']
30 static Config: z<Config> = z.object({
31 maxAudioBytes: z.natural().min(46).default(4 * 1024 * 1024),
32 maxDurationSeconds: z.number().min(1).default(120),
33 })
34
35 constructor(ctx: Context, private readonly config: Config) {
36 super(ctx, 'speechController', { namespace: 'speech' })
37 }
38
39 /**
40 * Read provider choices without preparing a recognizer.
41 * @returns available providers, resolved default, and recording limits.
42 */
43 @Remote
44 catalog(): SpeechCatalog {
45 return { ...this.ctx.speechToText.snapshot(), ...this.config }
46 }
47
48 /**
49 * Follow provider readiness independently of Session and preparation lifetimes.
50 * @param signal - Client observation lifetime.
51 * @returns initial and subsequent complete readiness snapshots.
52 */
53 @Remote({ mode: 'stream' })
54 async *follow(signal: AbortSignal): AsyncIterable<SpeechCatalog> {
55 for await (const snapshot of this.ctx.speechToText.follow(signal)) yield { ...snapshot, ...this.config }
56 }
57
58 /**
59 * Persist the user's recognition preferences.
60 * @param patch - changed preference fields.
61 * @returns after preferences are saved.
62 */
63 @Remote
64 configure(patch: SpeechSelectionPatch): Promise<void> { return this.ctx.speechToText.configure(patch) }
65
66 /**
67 * Start or join one Host-owned preparation task.
68 * @param providerId - selected recognizer.
69 * @param options - task-local source selection validated by the provider.
70 */
71 @Remote
72 prepare(providerId: SpeechProviderId, options?: SpeechPreparationOptions): void { this.ctx.speechToText.prepare(providerId, options) }
73
74 /**
75 * Explicitly cancel resource preparation.
76 * @param providerId - selected recognizer.
77 * @returns after the preparation task settles.
78 */
79 @Remote
80 cancelPreparation(providerId: SpeechProviderId): Promise<void> { return this.ctx.speechToText.cancelPreparation(providerId) }
81
82 /**
83 * Validate and transcribe one recording through the explicit provider selection.
84 * @param request - canonical WAV encoded as base64, provider id and language hint.
85 * @param signal - Client cancellation or Remote contribution disposal.
86 * @returns final transcript without adding a Session event.
87 */
88 @Remote
89 async transcribe(request: TranscriptionRequest, signal: AbortSignal): Promise<Transcript> {
90 signal.throwIfAborted()
91 const encoded = request.audioBase64
92 if (encoded.length > Math.ceil(this.config.maxAudioBytes / 3) * 4) {
93 throw new RemoteError('speech/invalid-audio', 'Audio is invalid or exceeds the configured byte limit', { reason: 'encoding-or-size' })
94 }
95 const audio = Buffer.from(encoded, 'base64')
96 try {
97 if (audio.toString('base64') !== encoded) throw new Error('Audio must use canonical base64 encoding')
98 if (audio.length > this.config.maxAudioBytes) throw new Error('Audio exceeds the configured byte limit')
99 validateWave(audio, this.config.maxDurationSeconds)
100 const spec = this.ctx.speechToText.resolve({ audio,
101 ...request.providerId === undefined ? {} : { providerId: request.providerId },
102 ...request.language === undefined ? {} : { language: request.language },
103 })
104 return await this.ctx.speechToText.transcribe(spec, signal)
105 } catch (error) {
106 signal.throwIfAborted()
107 const reason = error instanceof Error ? error.message : String(error)
108 throw new RemoteError('speech/transcription-failed', reason, { reason })
109 }
110 }
111}