mirror of
https://github.com/hansjone/oclaw.git
synced 2026-10-10 11:50:45 +08:00
重构主控编排与运行时预热链路,统一工作区提示词/专家调度协议并补齐 wiki 记忆注入与写回闭环。
同时收敛启动与运维脚本默认行为(含 wiki worker)、更新 Admin 可观测性与相关测试,降低首轮时延并提高运行稳定性。 Made-with: Cursor
This commit is contained in:
parent
4a23b715a2
commit
dbbe3add6a
14438 changed files with 2693620 additions and 2546 deletions
11
openclaw/extensions/microsoft/index.ts
Normal file
11
openclaw/extensions/microsoft/index.ts
Normal file
|
|
@ -0,0 +1,11 @@
|
|||
import { definePluginEntry } from "openclaw/plugin-sdk/plugin-entry";
|
||||
import { buildMicrosoftSpeechProvider } from "./speech-provider.js";
|
||||
|
||||
export default definePluginEntry({
|
||||
id: "microsoft",
|
||||
name: "Microsoft Speech",
|
||||
description: "Bundled Microsoft speech provider",
|
||||
register(api) {
|
||||
api.registerSpeechProvider(buildMicrosoftSpeechProvider());
|
||||
},
|
||||
});
|
||||
12
openclaw/extensions/microsoft/openclaw.plugin.json
Normal file
12
openclaw/extensions/microsoft/openclaw.plugin.json
Normal file
|
|
@ -0,0 +1,12 @@
|
|||
{
|
||||
"id": "microsoft",
|
||||
"enabledByDefault": true,
|
||||
"contracts": {
|
||||
"speechProviders": ["microsoft"]
|
||||
},
|
||||
"configSchema": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"properties": {}
|
||||
}
|
||||
}
|
||||
18
openclaw/extensions/microsoft/package.json
Normal file
18
openclaw/extensions/microsoft/package.json
Normal file
|
|
@ -0,0 +1,18 @@
|
|||
{
|
||||
"name": "@openclaw/microsoft-speech",
|
||||
"version": "2026.4.20",
|
||||
"private": true,
|
||||
"description": "OpenClaw Microsoft speech plugin",
|
||||
"type": "module",
|
||||
"dependencies": {
|
||||
"node-edge-tts": "^1.2.10"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@openclaw/plugin-sdk": "workspace:*"
|
||||
},
|
||||
"openclaw": {
|
||||
"extensions": [
|
||||
"./index.ts"
|
||||
]
|
||||
}
|
||||
}
|
||||
267
openclaw/extensions/microsoft/speech-provider.test.ts
Normal file
267
openclaw/extensions/microsoft/speech-provider.test.ts
Normal file
|
|
@ -0,0 +1,267 @@
|
|||
import { mkdtempSync, writeFileSync } from "node:fs";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
import type { OpenClawConfig } from "openclaw/plugin-sdk/config-runtime";
|
||||
import { afterEach, describe, expect, it, vi } from "vitest";
|
||||
import {
|
||||
buildMicrosoftSpeechProvider,
|
||||
isCjkDominant,
|
||||
listMicrosoftVoices,
|
||||
} from "./speech-provider.js";
|
||||
import * as ttsModule from "./tts.js";
|
||||
|
||||
const TEST_CFG = {} as OpenClawConfig;
|
||||
|
||||
describe("listMicrosoftVoices", () => {
|
||||
const originalFetch = globalThis.fetch;
|
||||
const proxyEnvKeys = [
|
||||
"OPENCLAW_DEBUG_PROXY_ENABLED",
|
||||
"OPENCLAW_DEBUG_PROXY_DB_PATH",
|
||||
"OPENCLAW_DEBUG_PROXY_BLOB_DIR",
|
||||
"OPENCLAW_DEBUG_PROXY_SESSION_ID",
|
||||
] as const;
|
||||
let priorProxyEnv: Partial<Record<(typeof proxyEnvKeys)[number], string | undefined>> = {};
|
||||
|
||||
afterEach(() => {
|
||||
globalThis.fetch = originalFetch;
|
||||
vi.restoreAllMocks();
|
||||
for (const key of proxyEnvKeys) {
|
||||
const value = priorProxyEnv[key];
|
||||
if (value === undefined) {
|
||||
delete process.env[key];
|
||||
} else {
|
||||
process.env[key] = value;
|
||||
}
|
||||
}
|
||||
priorProxyEnv = {};
|
||||
});
|
||||
|
||||
it("maps Microsoft voice metadata into speech voice options", async () => {
|
||||
globalThis.fetch = vi.fn().mockResolvedValue(
|
||||
new Response(
|
||||
JSON.stringify([
|
||||
{
|
||||
ShortName: "en-US-AvaNeural",
|
||||
FriendlyName: "Microsoft Ava Online (Natural) - English (United States)",
|
||||
Locale: "en-US",
|
||||
Gender: "Female",
|
||||
VoiceTag: {
|
||||
ContentCategories: ["General"],
|
||||
VoicePersonalities: ["Friendly", "Positive"],
|
||||
},
|
||||
},
|
||||
]),
|
||||
{ status: 200 },
|
||||
),
|
||||
) as unknown as typeof globalThis.fetch;
|
||||
|
||||
const voices = await listMicrosoftVoices();
|
||||
|
||||
expect(voices).toEqual([
|
||||
{
|
||||
id: "en-US-AvaNeural",
|
||||
name: "Microsoft Ava Online (Natural) - English (United States)",
|
||||
category: "General",
|
||||
description: "Friendly, Positive",
|
||||
locale: "en-US",
|
||||
gender: "Female",
|
||||
personalities: ["Friendly", "Positive"],
|
||||
},
|
||||
]);
|
||||
});
|
||||
|
||||
it("throws on Microsoft voice list failures", async () => {
|
||||
globalThis.fetch = vi
|
||||
.fn()
|
||||
.mockResolvedValue(
|
||||
new Response("nope", { status: 503 }),
|
||||
) as unknown as typeof globalThis.fetch;
|
||||
|
||||
await expect(listMicrosoftVoices()).rejects.toThrow("Microsoft voices API error (503)");
|
||||
});
|
||||
|
||||
it("records voice discovery exchanges in debug proxy capture mode", async () => {
|
||||
const tempDir = mkdtempSync(path.join(os.tmpdir(), "microsoft-voices-capture-"));
|
||||
priorProxyEnv = Object.fromEntries(
|
||||
proxyEnvKeys.map((key) => [key, process.env[key]]),
|
||||
) as typeof priorProxyEnv;
|
||||
process.env.OPENCLAW_DEBUG_PROXY_ENABLED = "1";
|
||||
process.env.OPENCLAW_DEBUG_PROXY_DB_PATH = path.join(tempDir, "capture.sqlite");
|
||||
process.env.OPENCLAW_DEBUG_PROXY_BLOB_DIR = path.join(tempDir, "blobs");
|
||||
process.env.OPENCLAW_DEBUG_PROXY_SESSION_ID = "ms-voices-session";
|
||||
|
||||
globalThis.fetch = vi
|
||||
.fn()
|
||||
.mockResolvedValue(
|
||||
new Response(JSON.stringify([{ ShortName: "en-US-AvaNeural" }]), { status: 200 }),
|
||||
) as unknown as typeof globalThis.fetch;
|
||||
|
||||
const { getDebugProxyCaptureStore } = await import("../../src/proxy-capture/store.sqlite.js");
|
||||
const store = getDebugProxyCaptureStore(
|
||||
process.env.OPENCLAW_DEBUG_PROXY_DB_PATH,
|
||||
process.env.OPENCLAW_DEBUG_PROXY_BLOB_DIR,
|
||||
);
|
||||
store.upsertSession({
|
||||
id: "ms-voices-session",
|
||||
startedAt: Date.now(),
|
||||
mode: "test",
|
||||
sourceScope: "openclaw",
|
||||
sourceProcess: "openclaw",
|
||||
dbPath: process.env.OPENCLAW_DEBUG_PROXY_DB_PATH,
|
||||
blobDir: process.env.OPENCLAW_DEBUG_PROXY_BLOB_DIR,
|
||||
});
|
||||
|
||||
await listMicrosoftVoices();
|
||||
await new Promise((resolve) => setTimeout(resolve, 0));
|
||||
|
||||
const events = store.getSessionEvents("ms-voices-session", 10);
|
||||
expect(
|
||||
events.some((event) => event.kind === "request" && event.host === "speech.platform.bing.com"),
|
||||
).toBe(true);
|
||||
expect(
|
||||
events.some(
|
||||
(event) => event.kind === "response" && event.host === "speech.platform.bing.com",
|
||||
),
|
||||
).toBe(true);
|
||||
});
|
||||
|
||||
it("does not double-capture voice discovery when the global fetch patch is installed", async () => {
|
||||
const tempDir = mkdtempSync(path.join(os.tmpdir(), "microsoft-voices-global-"));
|
||||
priorProxyEnv = Object.fromEntries(
|
||||
proxyEnvKeys.map((key) => [key, process.env[key]]),
|
||||
) as typeof priorProxyEnv;
|
||||
process.env.OPENCLAW_DEBUG_PROXY_ENABLED = "1";
|
||||
process.env.OPENCLAW_DEBUG_PROXY_DB_PATH = path.join(tempDir, "capture.sqlite");
|
||||
process.env.OPENCLAW_DEBUG_PROXY_BLOB_DIR = path.join(tempDir, "blobs");
|
||||
process.env.OPENCLAW_DEBUG_PROXY_SESSION_ID = "ms-voices-global-session";
|
||||
|
||||
globalThis.fetch = vi.fn(
|
||||
async () => new Response(JSON.stringify([{ ShortName: "en-US-AvaNeural" }]), { status: 200 }),
|
||||
) as unknown as typeof globalThis.fetch;
|
||||
|
||||
const { getDebugProxyCaptureStore } = await import("../../src/proxy-capture/store.sqlite.js");
|
||||
const { finalizeDebugProxyCapture, initializeDebugProxyCapture } =
|
||||
await import("../../src/proxy-capture/runtime.js");
|
||||
const store = getDebugProxyCaptureStore(
|
||||
process.env.OPENCLAW_DEBUG_PROXY_DB_PATH,
|
||||
process.env.OPENCLAW_DEBUG_PROXY_BLOB_DIR,
|
||||
);
|
||||
store.upsertSession({
|
||||
id: "ms-voices-global-session",
|
||||
startedAt: Date.now(),
|
||||
mode: "test",
|
||||
sourceScope: "openclaw",
|
||||
sourceProcess: "openclaw",
|
||||
dbPath: process.env.OPENCLAW_DEBUG_PROXY_DB_PATH,
|
||||
blobDir: process.env.OPENCLAW_DEBUG_PROXY_BLOB_DIR,
|
||||
});
|
||||
initializeDebugProxyCapture("test");
|
||||
|
||||
try {
|
||||
await listMicrosoftVoices();
|
||||
await new Promise((resolve) => setTimeout(resolve, 0));
|
||||
|
||||
const events = store
|
||||
.getSessionEvents("ms-voices-global-session", 10)
|
||||
.filter((event) => event.host === "speech.platform.bing.com");
|
||||
expect(events).toHaveLength(2);
|
||||
const kinds = events.map((event) => String(event.kind)).toSorted();
|
||||
expect(kinds).toEqual(["request", "response"]);
|
||||
} finally {
|
||||
globalThis.fetch = originalFetch;
|
||||
finalizeDebugProxyCapture();
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe("isCjkDominant", () => {
|
||||
it("returns true for Chinese text", () => {
|
||||
expect(isCjkDominant("你好世界")).toBe(true);
|
||||
});
|
||||
|
||||
it("returns true for mixed text with majority CJK", () => {
|
||||
expect(isCjkDominant("你好,这是一个测试 hello")).toBe(true);
|
||||
});
|
||||
|
||||
it("returns false for English text", () => {
|
||||
expect(isCjkDominant("Hello, this is a test")).toBe(false);
|
||||
});
|
||||
|
||||
it("returns false for empty string", () => {
|
||||
expect(isCjkDominant("")).toBe(false);
|
||||
});
|
||||
|
||||
it("returns false for mostly English with a few CJK chars", () => {
|
||||
expect(isCjkDominant("This is a long English sentence with one 字")).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("buildMicrosoftSpeechProvider", () => {
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
it("switches to a Chinese voice for CJK text when no explicit voice override is set", async () => {
|
||||
const provider = buildMicrosoftSpeechProvider();
|
||||
const edgeSpy = vi.spyOn(ttsModule, "edgeTTS").mockImplementation(async ({ outputPath }) => {
|
||||
writeFileSync(outputPath, Buffer.from([0xff, 0xfb, 0x90, 0x00]));
|
||||
});
|
||||
|
||||
await provider.synthesize({
|
||||
text: "你好,这是一个测试 hello",
|
||||
cfg: TEST_CFG,
|
||||
providerConfig: {
|
||||
enabled: true,
|
||||
voice: "en-US-MichelleNeural",
|
||||
lang: "en-US",
|
||||
outputFormat: "audio-24khz-48kbitrate-mono-mp3",
|
||||
outputFormatConfigured: true,
|
||||
saveSubtitles: false,
|
||||
},
|
||||
providerOverrides: {},
|
||||
timeoutMs: 1000,
|
||||
target: "audio-file",
|
||||
});
|
||||
|
||||
expect(edgeSpy).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
config: expect.objectContaining({
|
||||
voice: "zh-CN-XiaoxiaoNeural",
|
||||
lang: "zh-CN",
|
||||
}),
|
||||
}),
|
||||
);
|
||||
});
|
||||
|
||||
it("preserves an explicitly configured English voice for CJK text", async () => {
|
||||
const provider = buildMicrosoftSpeechProvider();
|
||||
const edgeSpy = vi.spyOn(ttsModule, "edgeTTS").mockImplementation(async ({ outputPath }) => {
|
||||
writeFileSync(outputPath, Buffer.from([0xff, 0xfb, 0x90, 0x00]));
|
||||
});
|
||||
|
||||
await provider.synthesize({
|
||||
text: "你好,这是一个测试 hello",
|
||||
cfg: TEST_CFG,
|
||||
providerConfig: {
|
||||
enabled: true,
|
||||
voice: "en-US-AvaNeural",
|
||||
lang: "en-US",
|
||||
outputFormat: "audio-24khz-48kbitrate-mono-mp3",
|
||||
outputFormatConfigured: true,
|
||||
saveSubtitles: false,
|
||||
},
|
||||
providerOverrides: {},
|
||||
timeoutMs: 1000,
|
||||
target: "audio-file",
|
||||
});
|
||||
|
||||
expect(edgeSpy).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
config: expect.objectContaining({
|
||||
voice: "en-US-AvaNeural",
|
||||
lang: "en-US",
|
||||
}),
|
||||
}),
|
||||
);
|
||||
});
|
||||
});
|
||||
281
openclaw/extensions/microsoft/speech-provider.ts
Normal file
281
openclaw/extensions/microsoft/speech-provider.ts
Normal file
|
|
@ -0,0 +1,281 @@
|
|||
import { mkdirSync, mkdtempSync, readFileSync, rmSync } from "node:fs";
|
||||
import path from "node:path";
|
||||
import {
|
||||
CHROMIUM_FULL_VERSION,
|
||||
TRUSTED_CLIENT_TOKEN,
|
||||
generateSecMsGecToken,
|
||||
} from "node-edge-tts/dist/drm.js";
|
||||
import { isVoiceCompatibleAudio } from "openclaw/plugin-sdk/media-runtime";
|
||||
import {
|
||||
captureHttpExchange,
|
||||
isDebugProxyGlobalFetchPatchInstalled,
|
||||
} from "openclaw/plugin-sdk/proxy-capture";
|
||||
import type {
|
||||
SpeechProviderConfig,
|
||||
SpeechProviderPlugin,
|
||||
SpeechVoiceOption,
|
||||
} from "openclaw/plugin-sdk/speech";
|
||||
import { asBoolean, asFiniteNumber, asObject, trimToUndefined } from "openclaw/plugin-sdk/speech";
|
||||
import { resolvePreferredOpenClawTmpDir } from "openclaw/plugin-sdk/temp-path";
|
||||
import { edgeTTS, inferEdgeExtension } from "./tts.js";
|
||||
|
||||
const DEFAULT_EDGE_VOICE = "en-US-MichelleNeural";
|
||||
const DEFAULT_EDGE_LANG = "en-US";
|
||||
const DEFAULT_EDGE_OUTPUT_FORMAT = "audio-24khz-48kbitrate-mono-mp3";
|
||||
|
||||
type MicrosoftProviderConfig = {
|
||||
enabled: boolean;
|
||||
voice: string;
|
||||
lang: string;
|
||||
outputFormat: string;
|
||||
outputFormatConfigured: boolean;
|
||||
pitch?: string;
|
||||
rate?: string;
|
||||
volume?: string;
|
||||
saveSubtitles: boolean;
|
||||
proxy?: string;
|
||||
timeoutMs?: number;
|
||||
};
|
||||
|
||||
type MicrosoftVoiceListEntry = {
|
||||
ShortName?: string;
|
||||
FriendlyName?: string;
|
||||
Locale?: string;
|
||||
Gender?: string;
|
||||
VoiceTag?: {
|
||||
ContentCategories?: string[];
|
||||
VoicePersonalities?: string[];
|
||||
};
|
||||
};
|
||||
|
||||
function normalizeMicrosoftProviderConfig(
|
||||
rawConfig: Record<string, unknown>,
|
||||
): MicrosoftProviderConfig {
|
||||
const providers = asObject(rawConfig.providers);
|
||||
const rawEdge = asObject(rawConfig.edge);
|
||||
const rawMicrosoft = asObject(rawConfig.microsoft);
|
||||
const rawProvider = asObject(providers?.microsoft);
|
||||
const raw = { ...rawEdge, ...rawMicrosoft, ...rawProvider };
|
||||
const outputFormat = trimToUndefined(raw.outputFormat);
|
||||
return {
|
||||
enabled: asBoolean(raw.enabled) ?? true,
|
||||
voice: trimToUndefined(raw.voice) ?? DEFAULT_EDGE_VOICE,
|
||||
lang: trimToUndefined(raw.lang) ?? DEFAULT_EDGE_LANG,
|
||||
outputFormat: outputFormat ?? DEFAULT_EDGE_OUTPUT_FORMAT,
|
||||
outputFormatConfigured: Boolean(outputFormat),
|
||||
pitch: trimToUndefined(raw.pitch),
|
||||
rate: trimToUndefined(raw.rate),
|
||||
volume: trimToUndefined(raw.volume),
|
||||
saveSubtitles: asBoolean(raw.saveSubtitles) ?? false,
|
||||
proxy: trimToUndefined(raw.proxy),
|
||||
timeoutMs: asFiniteNumber(raw.timeoutMs),
|
||||
};
|
||||
}
|
||||
|
||||
function readMicrosoftProviderConfig(config: SpeechProviderConfig): MicrosoftProviderConfig {
|
||||
const defaults = normalizeMicrosoftProviderConfig({});
|
||||
return {
|
||||
enabled: asBoolean(config.enabled) ?? defaults.enabled,
|
||||
voice: trimToUndefined(config.voice) ?? defaults.voice,
|
||||
lang: trimToUndefined(config.lang) ?? defaults.lang,
|
||||
outputFormat: trimToUndefined(config.outputFormat) ?? defaults.outputFormat,
|
||||
outputFormatConfigured:
|
||||
asBoolean(config.outputFormatConfigured) ?? defaults.outputFormatConfigured,
|
||||
pitch: trimToUndefined(config.pitch) ?? defaults.pitch,
|
||||
rate: trimToUndefined(config.rate) ?? defaults.rate,
|
||||
volume: trimToUndefined(config.volume) ?? defaults.volume,
|
||||
saveSubtitles: asBoolean(config.saveSubtitles) ?? defaults.saveSubtitles,
|
||||
proxy: trimToUndefined(config.proxy) ?? defaults.proxy,
|
||||
timeoutMs: asFiniteNumber(config.timeoutMs) ?? defaults.timeoutMs,
|
||||
};
|
||||
}
|
||||
|
||||
function buildMicrosoftVoiceHeaders(): Record<string, string> {
|
||||
const major = CHROMIUM_FULL_VERSION.split(".")[0] || "0";
|
||||
return {
|
||||
Authority: "speech.platform.bing.com",
|
||||
Origin: "chrome-extension://jdiccldimpdaibmpdkjnbmckianbfold",
|
||||
Accept: "*/*",
|
||||
"User-Agent":
|
||||
`Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 ` +
|
||||
`(KHTML, like Gecko) Chrome/${major}.0.0.0 Safari/537.36 Edg/${major}.0.0.0`,
|
||||
"Sec-MS-GEC": generateSecMsGecToken(),
|
||||
"Sec-MS-GEC-Version": `1-${CHROMIUM_FULL_VERSION}`,
|
||||
};
|
||||
}
|
||||
|
||||
function formatMicrosoftVoiceDescription(entry: MicrosoftVoiceListEntry): string | undefined {
|
||||
const personalities = entry.VoiceTag?.VoicePersonalities?.filter(Boolean) ?? [];
|
||||
return personalities.length > 0 ? personalities.join(", ") : undefined;
|
||||
}
|
||||
|
||||
export function isCjkDominant(text: string): boolean {
|
||||
const stripped = text.replace(/\s+/g, "");
|
||||
if (stripped.length === 0) {
|
||||
return false;
|
||||
}
|
||||
let cjkCount = 0;
|
||||
for (const ch of stripped) {
|
||||
const code = ch.codePointAt(0) ?? 0;
|
||||
if (
|
||||
(code >= 0x4e00 && code <= 0x9fff) ||
|
||||
(code >= 0x3400 && code <= 0x4dbf) ||
|
||||
(code >= 0x3000 && code <= 0x303f) ||
|
||||
(code >= 0xff00 && code <= 0xffef)
|
||||
) {
|
||||
cjkCount += 1;
|
||||
}
|
||||
}
|
||||
return cjkCount / stripped.length > 0.3;
|
||||
}
|
||||
|
||||
const DEFAULT_CHINESE_EDGE_VOICE = "zh-CN-XiaoxiaoNeural";
|
||||
const DEFAULT_CHINESE_EDGE_LANG = "zh-CN";
|
||||
|
||||
export async function listMicrosoftVoices(): Promise<SpeechVoiceOption[]> {
|
||||
const url =
|
||||
"https://speech.platform.bing.com/consumer/speech/synthesize/readaloud/voices/list" +
|
||||
`?trustedclienttoken=${TRUSTED_CLIENT_TOKEN}`;
|
||||
const headers = buildMicrosoftVoiceHeaders();
|
||||
const response = await fetch(url, {
|
||||
headers,
|
||||
});
|
||||
if (!isDebugProxyGlobalFetchPatchInstalled()) {
|
||||
captureHttpExchange({
|
||||
url,
|
||||
method: "GET",
|
||||
requestHeaders: headers,
|
||||
response,
|
||||
transport: "http",
|
||||
meta: {
|
||||
provider: "microsoft",
|
||||
capability: "speech-voices",
|
||||
},
|
||||
});
|
||||
}
|
||||
if (!response.ok) {
|
||||
throw new Error(`Microsoft voices API error (${response.status})`);
|
||||
}
|
||||
const voices = (await response.json()) as MicrosoftVoiceListEntry[];
|
||||
return Array.isArray(voices)
|
||||
? voices
|
||||
.map((voice) => ({
|
||||
id: voice.ShortName?.trim() ?? "",
|
||||
name: trimToUndefined(voice.FriendlyName) ?? trimToUndefined(voice.ShortName),
|
||||
category: voice.VoiceTag?.ContentCategories?.find((value) => value.trim().length > 0),
|
||||
description: formatMicrosoftVoiceDescription(voice),
|
||||
locale: trimToUndefined(voice.Locale),
|
||||
gender: trimToUndefined(voice.Gender),
|
||||
personalities: voice.VoiceTag?.VoicePersonalities?.filter(
|
||||
(value): value is string => value.trim().length > 0,
|
||||
),
|
||||
}))
|
||||
.filter((voice) => voice.id.length > 0)
|
||||
: [];
|
||||
}
|
||||
|
||||
export function buildMicrosoftSpeechProvider(): SpeechProviderPlugin {
|
||||
return {
|
||||
id: "microsoft",
|
||||
label: "Microsoft",
|
||||
aliases: ["edge"],
|
||||
autoSelectOrder: 30,
|
||||
resolveConfig: ({ rawConfig }) => normalizeMicrosoftProviderConfig(rawConfig),
|
||||
resolveTalkConfig: ({ baseTtsConfig, talkProviderConfig }) => {
|
||||
const base = normalizeMicrosoftProviderConfig(baseTtsConfig);
|
||||
return {
|
||||
...base,
|
||||
enabled: true,
|
||||
...(trimToUndefined(talkProviderConfig.voiceId) == null
|
||||
? {}
|
||||
: { voice: trimToUndefined(talkProviderConfig.voiceId) }),
|
||||
...(trimToUndefined(talkProviderConfig.languageCode) == null
|
||||
? {}
|
||||
: { lang: trimToUndefined(talkProviderConfig.languageCode) }),
|
||||
...(trimToUndefined(talkProviderConfig.outputFormat) == null
|
||||
? {}
|
||||
: { outputFormat: trimToUndefined(talkProviderConfig.outputFormat) }),
|
||||
...(trimToUndefined(talkProviderConfig.pitch) == null
|
||||
? {}
|
||||
: { pitch: trimToUndefined(talkProviderConfig.pitch) }),
|
||||
...(trimToUndefined(talkProviderConfig.rate) == null
|
||||
? {}
|
||||
: { rate: trimToUndefined(talkProviderConfig.rate) }),
|
||||
...(trimToUndefined(talkProviderConfig.volume) == null
|
||||
? {}
|
||||
: { volume: trimToUndefined(talkProviderConfig.volume) }),
|
||||
...(trimToUndefined(talkProviderConfig.proxy) == null
|
||||
? {}
|
||||
: { proxy: trimToUndefined(talkProviderConfig.proxy) }),
|
||||
...(asFiniteNumber(talkProviderConfig.timeoutMs) == null
|
||||
? {}
|
||||
: { timeoutMs: asFiniteNumber(talkProviderConfig.timeoutMs) }),
|
||||
};
|
||||
},
|
||||
resolveTalkOverrides: ({ params }) => ({
|
||||
...(trimToUndefined(params.voiceId) == null
|
||||
? {}
|
||||
: { voice: trimToUndefined(params.voiceId) }),
|
||||
...(trimToUndefined(params.outputFormat) == null
|
||||
? {}
|
||||
: { outputFormat: trimToUndefined(params.outputFormat) }),
|
||||
}),
|
||||
listVoices: async () => await listMicrosoftVoices(),
|
||||
isConfigured: ({ providerConfig }) => readMicrosoftProviderConfig(providerConfig).enabled,
|
||||
synthesize: async (req) => {
|
||||
const config = readMicrosoftProviderConfig(req.providerConfig);
|
||||
const tempRoot = resolvePreferredOpenClawTmpDir();
|
||||
mkdirSync(tempRoot, { recursive: true, mode: 0o700 });
|
||||
const tempDir = mkdtempSync(path.join(tempRoot, "tts-microsoft-"));
|
||||
const overrideVoice = trimToUndefined(req.providerOverrides?.voice);
|
||||
let voice = overrideVoice ?? config.voice;
|
||||
let lang = config.lang;
|
||||
let outputFormat =
|
||||
trimToUndefined(req.providerOverrides?.outputFormat) ?? config.outputFormat;
|
||||
const fallbackOutputFormat =
|
||||
outputFormat !== DEFAULT_EDGE_OUTPUT_FORMAT ? DEFAULT_EDGE_OUTPUT_FORMAT : undefined;
|
||||
|
||||
if (!overrideVoice && voice === DEFAULT_EDGE_VOICE && isCjkDominant(req.text)) {
|
||||
voice = DEFAULT_CHINESE_EDGE_VOICE;
|
||||
lang = DEFAULT_CHINESE_EDGE_LANG;
|
||||
}
|
||||
|
||||
try {
|
||||
const runEdge = async (format: string) => {
|
||||
const fileExtension = inferEdgeExtension(format);
|
||||
const outputPath = path.join(tempDir, `speech${fileExtension}`);
|
||||
await edgeTTS({
|
||||
text: req.text,
|
||||
outputPath,
|
||||
config: {
|
||||
...config,
|
||||
voice,
|
||||
lang,
|
||||
outputFormat: format,
|
||||
},
|
||||
timeoutMs: req.timeoutMs,
|
||||
});
|
||||
const audioBuffer = readFileSync(outputPath);
|
||||
return {
|
||||
audioBuffer,
|
||||
outputFormat: format,
|
||||
fileExtension,
|
||||
voiceCompatible: isVoiceCompatibleAudio({ fileName: outputPath }),
|
||||
};
|
||||
};
|
||||
|
||||
try {
|
||||
return await runEdge(outputFormat);
|
||||
} catch (error) {
|
||||
if (!fallbackOutputFormat || fallbackOutputFormat === outputFormat) {
|
||||
throw error;
|
||||
}
|
||||
outputFormat = fallbackOutputFormat;
|
||||
return await runEdge(outputFormat);
|
||||
}
|
||||
} finally {
|
||||
rmSync(tempDir, { recursive: true, force: true });
|
||||
}
|
||||
},
|
||||
};
|
||||
}
|
||||
1
openclaw/extensions/microsoft/test-api.ts
Normal file
1
openclaw/extensions/microsoft/test-api.ts
Normal file
|
|
@ -0,0 +1 @@
|
|||
export { buildMicrosoftSpeechProvider } from "./speech-provider.js";
|
||||
16
openclaw/extensions/microsoft/tsconfig.json
Normal file
16
openclaw/extensions/microsoft/tsconfig.json
Normal file
|
|
@ -0,0 +1,16 @@
|
|||
{
|
||||
"extends": "../tsconfig.package-boundary.base.json",
|
||||
"compilerOptions": {
|
||||
"rootDir": "."
|
||||
},
|
||||
"include": ["./*.ts", "./src/**/*.ts"],
|
||||
"exclude": [
|
||||
"./**/*.test.ts",
|
||||
"./dist/**",
|
||||
"./node_modules/**",
|
||||
"./src/test-support/**",
|
||||
"./src/**/*test-helpers.ts",
|
||||
"./src/**/*test-harness.ts",
|
||||
"./src/**/*test-support.ts"
|
||||
]
|
||||
}
|
||||
74
openclaw/extensions/microsoft/tts.test.ts
Normal file
74
openclaw/extensions/microsoft/tts.test.ts
Normal file
|
|
@ -0,0 +1,74 @@
|
|||
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
import path from "node:path";
|
||||
import { afterEach, beforeAll, describe, expect, it, vi } from "vitest";
|
||||
|
||||
let edgeTTS: typeof import("./tts.js").edgeTTS;
|
||||
|
||||
let mockTtsPromise = vi.fn<(text: string, filePath: string) => Promise<void>>();
|
||||
|
||||
vi.mock("node-edge-tts", () => ({
|
||||
EdgeTTS: class {
|
||||
ttsPromise(text: string, filePath: string) {
|
||||
return mockTtsPromise(text, filePath);
|
||||
}
|
||||
},
|
||||
}));
|
||||
|
||||
const baseEdgeConfig = {
|
||||
voice: "en-US-MichelleNeural",
|
||||
lang: "en-US",
|
||||
outputFormat: "audio-24khz-48kbitrate-mono-mp3",
|
||||
saveSubtitles: false,
|
||||
};
|
||||
|
||||
describe("edgeTTS empty audio validation", () => {
|
||||
let tempDir: string | undefined;
|
||||
|
||||
beforeAll(async () => {
|
||||
({ edgeTTS } = await import("./tts.js"));
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
if (tempDir) {
|
||||
rmSync(tempDir, { recursive: true, force: true });
|
||||
tempDir = undefined;
|
||||
}
|
||||
});
|
||||
|
||||
it("throws when the output file is 0 bytes", async () => {
|
||||
tempDir = mkdtempSync(path.join(tmpdir(), "tts-test-"));
|
||||
const outputPath = path.join(tempDir, "voice.mp3");
|
||||
|
||||
mockTtsPromise = vi.fn(async (_text: string, filePath: string) => {
|
||||
writeFileSync(filePath, "");
|
||||
});
|
||||
|
||||
await expect(
|
||||
edgeTTS({
|
||||
text: "Hello",
|
||||
outputPath,
|
||||
config: baseEdgeConfig,
|
||||
timeoutMs: 10000,
|
||||
}),
|
||||
).rejects.toThrow("Edge TTS produced empty audio file");
|
||||
});
|
||||
|
||||
it("succeeds when the output file has content", async () => {
|
||||
tempDir = mkdtempSync(path.join(tmpdir(), "tts-test-"));
|
||||
const outputPath = path.join(tempDir, "voice.mp3");
|
||||
|
||||
mockTtsPromise = vi.fn(async (_text: string, filePath: string) => {
|
||||
writeFileSync(filePath, Buffer.from([0xff, 0xfb, 0x90, 0x00]));
|
||||
});
|
||||
|
||||
await expect(
|
||||
edgeTTS({
|
||||
text: "Hello",
|
||||
outputPath,
|
||||
config: baseEdgeConfig,
|
||||
timeoutMs: 10000,
|
||||
}),
|
||||
).resolves.toBeUndefined();
|
||||
});
|
||||
});
|
||||
56
openclaw/extensions/microsoft/tts.ts
Normal file
56
openclaw/extensions/microsoft/tts.ts
Normal file
|
|
@ -0,0 +1,56 @@
|
|||
import { statSync } from "node:fs";
|
||||
import { EdgeTTS } from "node-edge-tts";
|
||||
import { normalizeLowercaseStringOrEmpty } from "openclaw/plugin-sdk/text-runtime";
|
||||
|
||||
export function inferEdgeExtension(outputFormat: string): string {
|
||||
const normalized = normalizeLowercaseStringOrEmpty(outputFormat);
|
||||
if (normalized.includes("webm")) {
|
||||
return ".webm";
|
||||
}
|
||||
if (normalized.includes("ogg")) {
|
||||
return ".ogg";
|
||||
}
|
||||
if (normalized.includes("opus")) {
|
||||
return ".opus";
|
||||
}
|
||||
if (normalized.includes("wav") || normalized.includes("riff") || normalized.includes("pcm")) {
|
||||
return ".wav";
|
||||
}
|
||||
return ".mp3";
|
||||
}
|
||||
|
||||
export async function edgeTTS(params: {
|
||||
text: string;
|
||||
outputPath: string;
|
||||
config: {
|
||||
voice: string;
|
||||
lang: string;
|
||||
outputFormat: string;
|
||||
saveSubtitles: boolean;
|
||||
proxy?: string;
|
||||
rate?: string;
|
||||
pitch?: string;
|
||||
volume?: string;
|
||||
timeoutMs?: number;
|
||||
};
|
||||
timeoutMs: number;
|
||||
}): Promise<void> {
|
||||
const { text, outputPath, config, timeoutMs } = params;
|
||||
const tts = new EdgeTTS({
|
||||
voice: config.voice,
|
||||
lang: config.lang,
|
||||
outputFormat: config.outputFormat,
|
||||
saveSubtitles: config.saveSubtitles,
|
||||
proxy: config.proxy,
|
||||
rate: config.rate,
|
||||
pitch: config.pitch,
|
||||
volume: config.volume,
|
||||
timeout: config.timeoutMs ?? timeoutMs,
|
||||
});
|
||||
await tts.ttsPromise(text, outputPath);
|
||||
|
||||
const { size } = statSync(outputPath);
|
||||
if (size === 0) {
|
||||
throw new Error("Edge TTS produced empty audio file");
|
||||
}
|
||||
}
|
||||
Loading…
Add table
Add a link
Reference in a new issue