mirror of
https://github.com/openclaw/openclaw.git
synced 2026-08-23 10:55:31 -06:00
fa03d9b913
* refactor: consolidate coercion helpers * fix: remove duplicate coercion imports * fix: preserve serialized coercion guard * chore: ratchet coercion helper carve-outs * fix(test): keep gauntlet subprocess startup lean * fix: preserve imported session timestamp semantics * fix: preserve catalog timestamp string semantics * chore: align plugin SDK surface ratchet * fix: preserve trajectory and SDK string contracts * fix(test): preserve QA record assertion semantics * fix: complete standalone record guard rename * refactor(cron): use canonical string coercion * fix(acpx): preserve Pi timestamp parsing * test(channels): adapt custody test harnesses * test(telegram): classify media harness as test support * test(acpx): split timestamp contract coverage * test(channels): support generated custody contracts * chore: ban the full coercion helper name set Extends the declaration guard to all eleven consolidated helper names and renames the cron schedule-identity readNumber wrapper to readScheduleInteger so the banned generic name cannot regrow. * fix(scripts): repair release-validation guard drift and lint cause Restores the renamed isJsonRecord guard in assertTrustedWorkflowHarness after main added isRecord call sites in parallel, and attaches the caught YAML error as the thrown error cause (preserve-caught-error was red on main). * fix: preserve Claude timestamp string semantics * fix: preserve persisted timestamp string semantics * fix: preserve date-first timestamp contracts * fix(openai): harden delegation failure formatting * chore: close coercion helper guard gaps * test(openai): model non-error delegation rejection * chore: refresh plugin SDK API contract * fix(tasks): use canonical string field reader * fix(ai): use canonical provider error field coercion * fix(browser): migrate native bootstrap coercion * docs(plugin-sdk): clarify text record export compatibility * fix(gateway): normalize approval execution identity * test(outbound): isolate message action poll harness
236 lines
8.1 KiB
TypeScript
236 lines
8.1 KiB
TypeScript
/**
|
|
* Azure Speech REST helpers. They normalize endpoints, build SSML, list voices,
|
|
* and synthesize speech with response-size and SSRF guards.
|
|
*/
|
|
import {
|
|
assertOkOrThrowProviderError,
|
|
readProviderBinaryResponse,
|
|
readProviderJsonResponse,
|
|
} from "openclaw/plugin-sdk/provider-http";
|
|
import type { SpeechVoiceOption } from "openclaw/plugin-sdk/speech-core";
|
|
import { trimToUndefined } from "openclaw/plugin-sdk/speech-core";
|
|
import {
|
|
fetchWithSsrFGuard,
|
|
ssrfPolicyFromHttpBaseUrlAllowedHostname,
|
|
} from "openclaw/plugin-sdk/ssrf-runtime";
|
|
import { asOptionalRecord } from "openclaw/plugin-sdk/string-coerce-runtime";
|
|
|
|
/** Default Azure Speech neural voice. */
|
|
export const DEFAULT_AZURE_SPEECH_VOICE = "en-US-JennyNeural";
|
|
/** Default Azure Speech language. */
|
|
export const DEFAULT_AZURE_SPEECH_LANG = "en-US";
|
|
/** Default full-audio output format. */
|
|
export const DEFAULT_AZURE_SPEECH_AUDIO_FORMAT = "audio-24khz-48kbitrate-mono-mp3";
|
|
/** Default voice-note output format. */
|
|
export const DEFAULT_AZURE_SPEECH_VOICE_NOTE_FORMAT = "ogg-24khz-16bit-mono-opus";
|
|
/** Default telephony output format. */
|
|
export const DEFAULT_AZURE_SPEECH_TELEPHONY_FORMAT = "raw-8khz-8bit-mono-mulaw";
|
|
const DEFAULT_AZURE_SPEECH_MAX_BYTES = 16 * 1024 * 1024;
|
|
// Voice discovery should fail boundedly instead of waiting forever when the
|
|
// Azure Speech voices endpoint accepts the connection but never responds.
|
|
const DEFAULT_AZURE_SPEECH_VOICE_LIST_TIMEOUT_MS = 30_000;
|
|
|
|
/** Resolve and normalize the Azure Speech base URL from endpoint or region. */
|
|
export function normalizeAzureSpeechBaseUrl(params: {
|
|
baseUrl?: string;
|
|
endpoint?: string;
|
|
region?: string;
|
|
}): string | undefined {
|
|
const configured = trimToUndefined(params.baseUrl) ?? trimToUndefined(params.endpoint);
|
|
if (configured) {
|
|
return configured.replace(/\/+$/, "").replace(/\/cognitiveservices\/v1$/i, "");
|
|
}
|
|
const region = trimToUndefined(params.region);
|
|
return region ? `https://${region}.tts.speech.microsoft.com` : undefined;
|
|
}
|
|
|
|
function azureSpeechUrl(params: {
|
|
baseUrl?: string;
|
|
endpoint?: string;
|
|
region?: string;
|
|
path: "/cognitiveservices/v1" | "/cognitiveservices/voices/list";
|
|
}): string {
|
|
const baseUrl = normalizeAzureSpeechBaseUrl(params);
|
|
if (!baseUrl) {
|
|
throw new Error("Azure Speech region or endpoint missing");
|
|
}
|
|
return `${baseUrl}${params.path}`;
|
|
}
|
|
|
|
function escapeXmlText(text: string): string {
|
|
return text.replace(/&/g, "&").replace(/</g, "<").replace(/>/g, ">");
|
|
}
|
|
|
|
function escapeXmlAttr(value: string): string {
|
|
return escapeXmlText(value).replace(/"/g, """).replace(/'/g, "'");
|
|
}
|
|
|
|
/** Build escaped SSML for one Azure Speech synthesis request. */
|
|
function buildAzureSpeechSsml(params: { text: string; voice: string; lang?: string }): string {
|
|
const lang = trimToUndefined(params.lang) ?? DEFAULT_AZURE_SPEECH_LANG;
|
|
return (
|
|
`<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" ` +
|
|
`xml:lang="${escapeXmlAttr(lang)}">` +
|
|
`<voice name="${escapeXmlAttr(params.voice)}">${escapeXmlText(params.text)}</voice>` +
|
|
`</speak>`
|
|
);
|
|
}
|
|
|
|
/** Infer the generated audio file extension from Azure output format. */
|
|
export function inferAzureSpeechFileExtension(outputFormat: string): string {
|
|
const normalized = outputFormat.toLowerCase();
|
|
if (normalized.includes("mp3")) {
|
|
return ".mp3";
|
|
}
|
|
if (normalized.startsWith("ogg-")) {
|
|
return ".ogg";
|
|
}
|
|
if (normalized.startsWith("webm-")) {
|
|
return ".webm";
|
|
}
|
|
if (normalized.startsWith("riff-")) {
|
|
return ".wav";
|
|
}
|
|
if (normalized.startsWith("raw-")) {
|
|
return ".pcm";
|
|
}
|
|
if (normalized.startsWith("amr-")) {
|
|
return ".amr";
|
|
}
|
|
return ".audio";
|
|
}
|
|
|
|
/** Return whether an Azure output format is voice-note compatible. */
|
|
export function isAzureSpeechVoiceCompatible(outputFormat: string): boolean {
|
|
const normalized = outputFormat.toLowerCase();
|
|
return normalized.startsWith("ogg-") && normalized.includes("opus");
|
|
}
|
|
|
|
function readAzureVoiceTagStrings(value: unknown): string[] | undefined {
|
|
return Array.isArray(value)
|
|
? value.filter((entry): entry is string => trimToUndefined(entry) !== undefined)
|
|
: undefined;
|
|
}
|
|
|
|
function formatVoiceDescription(
|
|
tailoredScenarios: string[] | undefined,
|
|
personalities: string[] | undefined,
|
|
): string | undefined {
|
|
const parts = [...(tailoredScenarios ?? []), ...(personalities ?? [])];
|
|
return parts.length > 0 ? parts.join(", ") : undefined;
|
|
}
|
|
|
|
function isDeprecatedVoice(entry: Record<string, unknown>): boolean {
|
|
if (entry.IsDeprecated === true) {
|
|
return true;
|
|
}
|
|
if (typeof entry.IsDeprecated === "string" && entry.IsDeprecated.toLowerCase() === "true") {
|
|
return true;
|
|
}
|
|
const status = trimToUndefined(entry.Status)?.toLowerCase();
|
|
return status === "deprecated" || status === "retired" || status === "disabled";
|
|
}
|
|
|
|
/** List non-deprecated voices from the Azure Speech voices API. */
|
|
export async function listAzureSpeechVoices(params: {
|
|
apiKey: string;
|
|
baseUrl?: string;
|
|
endpoint?: string;
|
|
region?: string;
|
|
timeoutMs?: number;
|
|
}): Promise<SpeechVoiceOption[]> {
|
|
const url = azureSpeechUrl({ ...params, path: "/cognitiveservices/voices/list" });
|
|
const { response, release } = await fetchWithSsrFGuard({
|
|
url,
|
|
init: {
|
|
method: "GET",
|
|
headers: {
|
|
"Ocp-Apim-Subscription-Key": params.apiKey,
|
|
},
|
|
},
|
|
timeoutMs: params.timeoutMs ?? DEFAULT_AZURE_SPEECH_VOICE_LIST_TIMEOUT_MS,
|
|
policy: ssrfPolicyFromHttpBaseUrlAllowedHostname(url),
|
|
auditContext: "azure-speech.voices",
|
|
});
|
|
|
|
try {
|
|
await assertOkOrThrowProviderError(response, "Azure Speech voices API error");
|
|
const voices = await readProviderJsonResponse<unknown>(response, "azure-speech.voices");
|
|
return Array.isArray(voices)
|
|
? voices.flatMap((value) => {
|
|
const voice = asOptionalRecord(value);
|
|
const id = trimToUndefined(voice?.ShortName);
|
|
if (!voice || !id || isDeprecatedVoice(voice)) {
|
|
return [];
|
|
}
|
|
const voiceTag = asOptionalRecord(voice.VoiceTag);
|
|
const tailoredScenarios = readAzureVoiceTagStrings(voiceTag?.TailoredScenarios);
|
|
const personalities = readAzureVoiceTagStrings(voiceTag?.VoicePersonalities);
|
|
return [
|
|
{
|
|
id,
|
|
name: trimToUndefined(voice.DisplayName) ?? trimToUndefined(voice.LocalName),
|
|
description: formatVoiceDescription(tailoredScenarios, personalities),
|
|
locale: trimToUndefined(voice.Locale),
|
|
gender: trimToUndefined(voice.Gender),
|
|
personalities,
|
|
},
|
|
];
|
|
})
|
|
: [];
|
|
} finally {
|
|
await release();
|
|
}
|
|
}
|
|
|
|
/** Synthesize text to audio bytes using Azure Speech TTS. */
|
|
export async function azureSpeechTTS(params: {
|
|
text: string;
|
|
apiKey: string;
|
|
baseUrl?: string;
|
|
endpoint?: string;
|
|
region?: string;
|
|
voice?: string;
|
|
lang?: string;
|
|
outputFormat?: string;
|
|
timeoutMs?: number;
|
|
maxBytes?: number;
|
|
}): Promise<Buffer> {
|
|
const voice = trimToUndefined(params.voice) ?? DEFAULT_AZURE_SPEECH_VOICE;
|
|
const outputFormat = trimToUndefined(params.outputFormat) ?? DEFAULT_AZURE_SPEECH_AUDIO_FORMAT;
|
|
const url = azureSpeechUrl({ ...params, path: "/cognitiveservices/v1" });
|
|
const { response, release } = await fetchWithSsrFGuard({
|
|
url,
|
|
init: {
|
|
method: "POST",
|
|
headers: {
|
|
"Content-Type": "application/ssml+xml",
|
|
"Ocp-Apim-Subscription-Key": params.apiKey,
|
|
"X-Microsoft-OutputFormat": outputFormat,
|
|
"User-Agent": "OpenClaw",
|
|
},
|
|
body: buildAzureSpeechSsml({
|
|
text: params.text,
|
|
voice,
|
|
lang: params.lang,
|
|
}),
|
|
},
|
|
timeoutMs: params.timeoutMs,
|
|
policy: ssrfPolicyFromHttpBaseUrlAllowedHostname(url),
|
|
auditContext: "azure-speech.tts",
|
|
});
|
|
|
|
try {
|
|
await assertOkOrThrowProviderError(response, "Azure Speech TTS API error");
|
|
return Buffer.from(
|
|
await readProviderBinaryResponse(response, "Azure Speech TTS API error", "audio", {
|
|
maxBytes: params.maxBytes ?? DEFAULT_AZURE_SPEECH_MAX_BYTES,
|
|
onOverflow: ({ maxBytes }) =>
|
|
new Error(`Azure Speech TTS audio response exceeds ${maxBytes} bytes`),
|
|
}),
|
|
);
|
|
} finally {
|
|
await release();
|
|
}
|
|
}
|