1
/** Authenticated, cancellation-aware Client access to the speech capability. */2
import { Context } from '@deepseek-ai/cordis'3
import z from '@deepseek-ai/schemastery'4
import { Remote, RemoteError, TypertRemoteService } from '@deepseek-ai/dsh-typert-protocol'5
import type {} from '@deepseek-ai/dsh-experimental-speech-to-text'6
import type { SpeechPreparationOptions, SpeechProviderId, SpeechSelectionPatch, Transcript } from '@deepseek-ai/dsh-experimental-speech-to-text/types'7
import type { SpeechCatalog, TranscriptionRequest } from './types.ts'8
import { validateWave } from '@deepseek-ai/dsh-experimental-speech-to-text/wave'10
export type * from './types.ts'12
declare module '@deepseek-ai/cordis' {13
interface Context {14
/** Experimental speech Remote controller. */15
speechController: SpeechController16
}17
}19
/** Limits applied before decoding or calling a provider. */20
export interface Config {21
/** Maximum decoded WAV bytes per request. */22
maxAudioBytes: number23
/** Maximum PCM recording duration in seconds. */24
maxDurationSeconds: number25
}27
/** Speech calls never activate or submit to an Agent. */28
export default class SpeechController extends TypertRemoteService {29
static inject = ['speechToText', 'typert']30
static Config: z<Config> = z.object({31
maxAudioBytes: z.natural().min(46).default(4 * 1024 * 1024),32
maxDurationSeconds: z.number().min(1).default(120),33
})35
constructor(ctx: Context, private readonly config: Config) {36
super(ctx, 'speechController', { namespace: 'speech' })37
}39
/**40
* Read provider choices without preparing a recognizer.41
* @returns available providers, resolved default, and recording limits.42
*/43
@Remote44
catalog(): SpeechCatalog {45
return { ...this.ctx.speechToText.snapshot(), ...this.config }46
}48
/**49
* Follow provider readiness independently of Session and preparation lifetimes.50
* @param signal - Client observation lifetime.51
* @returns initial and subsequent complete readiness snapshots.52
*/53
@Remote({ mode: 'stream' })54
async *follow(signal: AbortSignal): AsyncIterable<SpeechCatalog> {55
for await (const snapshot of this.ctx.speechToText.follow(signal)) yield { ...snapshot, ...this.config }56
}58
/**59
* Persist the user's recognition preferences.60
* @param patch - changed preference fields.61
* @returns after preferences are saved.62
*/63
@Remote64
configure(patch: SpeechSelectionPatch): Promise<void> { return this.ctx.speechToText.configure(patch) }66
/**67
* Start or join one Host-owned preparation task.68
* @param providerId - selected recognizer.69
* @param options - task-local source selection validated by the provider.70
*/71
@Remote72
prepare(providerId: SpeechProviderId, options?: SpeechPreparationOptions): void { this.ctx.speechToText.prepare(providerId, options) }74
/**75
* Explicitly cancel resource preparation.76
* @param providerId - selected recognizer.77
* @returns after the preparation task settles.78
*/79
@Remote80
cancelPreparation(providerId: SpeechProviderId): Promise<void> { return this.ctx.speechToText.cancelPreparation(providerId) }82
/**83
* Validate and transcribe one recording through the explicit provider selection.84
* @param request - canonical WAV encoded as base64, provider id and language hint.85
* @param signal - Client cancellation or Remote contribution disposal.86
* @returns final transcript without adding a Session event.87
*/88
@Remote89
async transcribe(request: TranscriptionRequest, signal: AbortSignal): Promise<Transcript> {90
signal.throwIfAborted()91
const encoded = request.audioBase6492
if (encoded.length > Math.ceil(this.config.maxAudioBytes / 3) * 4) {93
throw new RemoteError('speech/invalid-audio', 'Audio is invalid or exceeds the configured byte limit', { reason: 'encoding-or-size' })94
}95
const audio = Buffer.from(encoded, 'base64')96
try {97
if (audio.toString('base64') !== encoded) throw new Error('Audio must use canonical base64 encoding')98
if (audio.length > this.config.maxAudioBytes) throw new Error('Audio exceeds the configured byte limit')99
validateWave(audio, this.config.maxDurationSeconds)100
const spec = this.ctx.speechToText.resolve({ audio,101
...request.providerId === undefined ? {} : { providerId: request.providerId },102
...request.language === undefined ? {} : { language: request.language },103
})104
return await this.ctx.speechToText.transcribe(spec, signal)105
} catch (error) {106
signal.throwIfAborted()107
const reason = error instanceof Error ? error.message : String(error)108
throw new RemoteError('speech/transcription-failed', reason, { reason })109
}110
}111
}