mirror of
https://github.com/hansjone/oclaw.git
synced 2026-10-11 15:03:14 +08:00
重构主控编排与运行时预热链路,统一工作区提示词/专家调度协议并补齐 wiki 记忆注入与写回闭环。
同时收敛启动与运维脚本默认行为(含 wiki worker)、更新 Admin 可观测性与相关测试,降低首轮时延并提高运行稳定性。 Made-with: Cursor
This commit is contained in:
parent
4a23b715a2
commit
dbbe3add6a
14438 changed files with 2693620 additions and 2546 deletions
16
openclaw/extensions/openai/api.ts
Normal file
16
openclaw/extensions/openai/api.ts
Normal file
|
|
@ -0,0 +1,16 @@
|
|||
export {
|
||||
applyOpenAIConfig,
|
||||
applyOpenAIProviderConfig,
|
||||
OPENAI_CODEX_DEFAULT_MODEL,
|
||||
OPENAI_DEFAULT_AUDIO_TRANSCRIPTION_MODEL,
|
||||
OPENAI_DEFAULT_EMBEDDING_MODEL,
|
||||
OPENAI_DEFAULT_IMAGE_MODEL,
|
||||
OPENAI_DEFAULT_MODEL,
|
||||
OPENAI_DEFAULT_TTS_MODEL,
|
||||
OPENAI_DEFAULT_TTS_VOICE,
|
||||
} from "./default-models.js";
|
||||
export { buildOpenAICodexProvider } from "./openai-codex-catalog.js";
|
||||
export { buildOpenAICodexProviderPlugin } from "./openai-codex-provider.js";
|
||||
export { buildOpenAIProvider } from "./openai-provider.js";
|
||||
export { buildOpenAIRealtimeTranscriptionProvider } from "./realtime-transcription-provider.js";
|
||||
export { buildOpenAIRealtimeVoiceProvider } from "./realtime-voice-provider.js";
|
||||
31
openclaw/extensions/openai/base-url.test.ts
Normal file
31
openclaw/extensions/openai/base-url.test.ts
Normal file
|
|
@ -0,0 +1,31 @@
|
|||
import { describe, expect, it } from "vitest";
|
||||
import { isOpenAIApiBaseUrl, isOpenAICodexBaseUrl } from "./base-url.js";
|
||||
|
||||
describe("openai base URL helpers", () => {
|
||||
it("recognizes direct OpenAI API routes", () => {
|
||||
expect(isOpenAIApiBaseUrl("https://api.openai.com")).toBe(true);
|
||||
expect(isOpenAIApiBaseUrl("https://api.openai.com/v1")).toBe(true);
|
||||
expect(isOpenAIApiBaseUrl("https://api.openai.com/v1/")).toBe(true);
|
||||
});
|
||||
|
||||
it("rejects proxy or unrelated API routes", () => {
|
||||
expect(isOpenAIApiBaseUrl("https://proxy.example.com/v1")).toBe(false);
|
||||
expect(isOpenAIApiBaseUrl("https://chatgpt.com/backend-api")).toBe(false);
|
||||
expect(isOpenAIApiBaseUrl(undefined)).toBe(false);
|
||||
});
|
||||
|
||||
it("recognizes Codex ChatGPT backend routes", () => {
|
||||
expect(isOpenAICodexBaseUrl("https://chatgpt.com/backend-api")).toBe(true);
|
||||
expect(isOpenAICodexBaseUrl("https://chatgpt.com/backend-api/")).toBe(true);
|
||||
expect(isOpenAICodexBaseUrl("https://chatgpt.com/backend-api/v1")).toBe(true);
|
||||
expect(isOpenAICodexBaseUrl("https://chatgpt.com/backend-api/v1/")).toBe(true);
|
||||
});
|
||||
|
||||
it("rejects non-Codex backend routes", () => {
|
||||
expect(isOpenAICodexBaseUrl("https://api.openai.com/v1")).toBe(false);
|
||||
expect(isOpenAICodexBaseUrl("https://chatgpt.com")).toBe(false);
|
||||
expect(isOpenAICodexBaseUrl("https://chatgpt.com/backend-api/v2")).toBe(false);
|
||||
expect(isOpenAICodexBaseUrl("https://chatgpt.com/backend-api/codex")).toBe(false);
|
||||
expect(isOpenAICodexBaseUrl(undefined)).toBe(false);
|
||||
});
|
||||
});
|
||||
17
openclaw/extensions/openai/base-url.ts
Normal file
17
openclaw/extensions/openai/base-url.ts
Normal file
|
|
@ -0,0 +1,17 @@
|
|||
import { normalizeOptionalString } from "openclaw/plugin-sdk/text-runtime";
|
||||
|
||||
export function isOpenAIApiBaseUrl(baseUrl?: string): boolean {
|
||||
const trimmed = normalizeOptionalString(baseUrl);
|
||||
if (!trimmed) {
|
||||
return false;
|
||||
}
|
||||
return /^https?:\/\/api\.openai\.com(?:\/v1)?\/?$/i.test(trimmed);
|
||||
}
|
||||
|
||||
export function isOpenAICodexBaseUrl(baseUrl?: string): boolean {
|
||||
const trimmed = normalizeOptionalString(baseUrl);
|
||||
if (!trimmed) {
|
||||
return false;
|
||||
}
|
||||
return /^https?:\/\/chatgpt\.com\/backend-api(?:\/v1)?\/?$/i.test(trimmed);
|
||||
}
|
||||
67
openclaw/extensions/openai/cli-backend.ts
Normal file
67
openclaw/extensions/openai/cli-backend.ts
Normal file
|
|
@ -0,0 +1,67 @@
|
|||
import type { CliBackendPlugin } from "openclaw/plugin-sdk/cli-backend";
|
||||
import {
|
||||
CLI_FRESH_WATCHDOG_DEFAULTS,
|
||||
CLI_RESUME_WATCHDOG_DEFAULTS,
|
||||
} from "openclaw/plugin-sdk/cli-backend";
|
||||
import { OPENAI_CODEX_DEFAULT_PROFILE_ID } from "./openai-codex-cli-auth.js";
|
||||
import { prepareOpenAICodexCliExecution } from "./openai-codex-cli-bridge.js";
|
||||
|
||||
const CODEX_CLI_DEFAULT_MODEL_REF = "codex-cli/gpt-5.4";
|
||||
|
||||
export function buildOpenAICodexCliBackend(): CliBackendPlugin {
|
||||
return {
|
||||
id: "codex-cli",
|
||||
liveTest: {
|
||||
defaultModelRef: CODEX_CLI_DEFAULT_MODEL_REF,
|
||||
defaultImageProbe: true,
|
||||
defaultMcpProbe: true,
|
||||
docker: {
|
||||
npmPackage: "@openai/codex",
|
||||
binaryName: "codex",
|
||||
},
|
||||
},
|
||||
bundleMcp: true,
|
||||
bundleMcpMode: "codex-config-overrides",
|
||||
defaultAuthProfileId: OPENAI_CODEX_DEFAULT_PROFILE_ID,
|
||||
authEpochMode: "profile-only",
|
||||
prepareExecution: prepareOpenAICodexCliExecution,
|
||||
config: {
|
||||
command: "codex",
|
||||
args: [
|
||||
"exec",
|
||||
"--json",
|
||||
"--color",
|
||||
"never",
|
||||
"--sandbox",
|
||||
"workspace-write",
|
||||
"--skip-git-repo-check",
|
||||
],
|
||||
resumeArgs: [
|
||||
"exec",
|
||||
"resume",
|
||||
"{sessionId}",
|
||||
"-c",
|
||||
'sandbox_mode="workspace-write"',
|
||||
"--skip-git-repo-check",
|
||||
],
|
||||
output: "jsonl",
|
||||
resumeOutput: "text",
|
||||
input: "arg",
|
||||
modelArg: "--model",
|
||||
sessionIdFields: ["thread_id"],
|
||||
sessionMode: "existing",
|
||||
systemPromptFileConfigArg: "-c",
|
||||
systemPromptFileConfigKey: "model_instructions_file",
|
||||
systemPromptWhen: "first",
|
||||
imageArg: "--image",
|
||||
imageMode: "repeat",
|
||||
reliability: {
|
||||
watchdog: {
|
||||
fresh: { ...CLI_FRESH_WATCHDOG_DEFAULTS },
|
||||
resume: { ...CLI_RESUME_WATCHDOG_DEFAULTS },
|
||||
},
|
||||
},
|
||||
serialize: true,
|
||||
},
|
||||
};
|
||||
}
|
||||
35
openclaw/extensions/openai/default-models.test.ts
Normal file
35
openclaw/extensions/openai/default-models.test.ts
Normal file
|
|
@ -0,0 +1,35 @@
|
|||
import type { OpenClawConfig } from "openclaw/plugin-sdk/provider-onboard";
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { applyOpenAIConfig, applyOpenAIProviderConfig, OPENAI_DEFAULT_MODEL } from "./api.js";
|
||||
|
||||
describe("openai default models", () => {
|
||||
it("adds allowlist entry for the default model", () => {
|
||||
const next = applyOpenAIProviderConfig({});
|
||||
expect(Object.keys(next.agents?.defaults?.models ?? {})).toContain(OPENAI_DEFAULT_MODEL);
|
||||
});
|
||||
|
||||
it("preserves existing alias for the default model", () => {
|
||||
const next = applyOpenAIProviderConfig({
|
||||
agents: {
|
||||
defaults: {
|
||||
models: {
|
||||
[OPENAI_DEFAULT_MODEL]: { alias: "My GPT" },
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
expect(next.agents?.defaults?.models?.[OPENAI_DEFAULT_MODEL]?.alias).toBe("My GPT");
|
||||
});
|
||||
|
||||
it("sets the default model when it is unset", () => {
|
||||
const next = applyOpenAIConfig({});
|
||||
expect(next.agents?.defaults?.model).toEqual({ primary: OPENAI_DEFAULT_MODEL });
|
||||
});
|
||||
|
||||
it("overrides model.primary while preserving fallbacks", () => {
|
||||
const next = applyOpenAIConfig({
|
||||
agents: { defaults: { model: { primary: "anthropic/claude-opus-4-6", fallbacks: [] } } },
|
||||
} as OpenClawConfig);
|
||||
expect(next.agents?.defaults?.model).toEqual({ primary: OPENAI_DEFAULT_MODEL, fallbacks: [] });
|
||||
});
|
||||
});
|
||||
40
openclaw/extensions/openai/default-models.ts
Normal file
40
openclaw/extensions/openai/default-models.ts
Normal file
|
|
@ -0,0 +1,40 @@
|
|||
import { ensureModelAllowlistEntry } from "openclaw/plugin-sdk/provider-onboard";
|
||||
import {
|
||||
applyAgentDefaultModelPrimary,
|
||||
type OpenClawConfig,
|
||||
} from "openclaw/plugin-sdk/provider-onboard";
|
||||
|
||||
export const OPENAI_DEFAULT_MODEL = "openai/gpt-5.4";
|
||||
export const OPENAI_CODEX_DEFAULT_MODEL = "openai-codex/gpt-5.4";
|
||||
export const OPENAI_DEFAULT_IMAGE_MODEL = "gpt-image-1";
|
||||
export const OPENAI_DEFAULT_TTS_MODEL = "gpt-4o-mini-tts";
|
||||
export const OPENAI_DEFAULT_TTS_VOICE = "alloy";
|
||||
export const OPENAI_DEFAULT_AUDIO_TRANSCRIPTION_MODEL = "gpt-4o-transcribe";
|
||||
export const OPENAI_DEFAULT_EMBEDDING_MODEL = "text-embedding-3-small";
|
||||
|
||||
export function applyOpenAIProviderConfig(cfg: OpenClawConfig): OpenClawConfig {
|
||||
const next = ensureModelAllowlistEntry({
|
||||
cfg,
|
||||
modelRef: OPENAI_DEFAULT_MODEL,
|
||||
});
|
||||
const models = { ...next.agents?.defaults?.models };
|
||||
models[OPENAI_DEFAULT_MODEL] = {
|
||||
...models[OPENAI_DEFAULT_MODEL],
|
||||
alias: models[OPENAI_DEFAULT_MODEL]?.alias ?? "GPT",
|
||||
};
|
||||
|
||||
return {
|
||||
...next,
|
||||
agents: {
|
||||
...next.agents,
|
||||
defaults: {
|
||||
...next.agents?.defaults,
|
||||
models,
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
export function applyOpenAIConfig(cfg: OpenClawConfig): OpenClawConfig {
|
||||
return applyAgentDefaultModelPrimary(applyOpenAIProviderConfig(cfg), OPENAI_DEFAULT_MODEL);
|
||||
}
|
||||
261
openclaw/extensions/openai/embedding-batch.ts
Normal file
261
openclaw/extensions/openai/embedding-batch.ts
Normal file
|
|
@ -0,0 +1,261 @@
|
|||
import {
|
||||
applyEmbeddingBatchOutputLine,
|
||||
buildBatchHeaders,
|
||||
buildEmbeddingBatchGroupOptions,
|
||||
EMBEDDING_BATCH_ENDPOINT,
|
||||
extractBatchErrorMessage,
|
||||
formatUnavailableBatchError,
|
||||
normalizeBatchBaseUrl,
|
||||
postJsonWithRetry,
|
||||
resolveBatchCompletionFromStatus,
|
||||
resolveCompletedBatchResult,
|
||||
runEmbeddingBatchGroups,
|
||||
throwIfBatchTerminalFailure,
|
||||
type EmbeddingBatchExecutionParams,
|
||||
type EmbeddingBatchStatus,
|
||||
type BatchCompletionResult,
|
||||
type ProviderBatchOutputLine,
|
||||
uploadBatchJsonlFile,
|
||||
withRemoteHttpResponse,
|
||||
} from "openclaw/plugin-sdk/memory-core-host-engine-embeddings";
|
||||
import type { OpenAiEmbeddingClient } from "./embedding-provider.js";
|
||||
|
||||
export type OpenAiBatchRequest = {
|
||||
custom_id: string;
|
||||
method: "POST";
|
||||
url: "/v1/embeddings";
|
||||
body: {
|
||||
model: string;
|
||||
input: string;
|
||||
};
|
||||
};
|
||||
|
||||
export type OpenAiBatchStatus = EmbeddingBatchStatus;
|
||||
export type OpenAiBatchOutputLine = ProviderBatchOutputLine;
|
||||
|
||||
export const OPENAI_BATCH_ENDPOINT = EMBEDDING_BATCH_ENDPOINT;
|
||||
const OPENAI_BATCH_COMPLETION_WINDOW = "24h";
|
||||
const OPENAI_BATCH_MAX_REQUESTS = 50000;
|
||||
|
||||
async function submitOpenAiBatch(params: {
|
||||
openAi: OpenAiEmbeddingClient;
|
||||
requests: OpenAiBatchRequest[];
|
||||
agentId: string;
|
||||
}): Promise<OpenAiBatchStatus> {
|
||||
const baseUrl = normalizeBatchBaseUrl(params.openAi);
|
||||
const inputFileId = await uploadBatchJsonlFile({
|
||||
client: params.openAi,
|
||||
requests: params.requests,
|
||||
errorPrefix: "openai batch file upload failed",
|
||||
});
|
||||
|
||||
return await postJsonWithRetry<OpenAiBatchStatus>({
|
||||
url: `${baseUrl}/batches`,
|
||||
headers: buildBatchHeaders(params.openAi, { json: true }),
|
||||
ssrfPolicy: params.openAi.ssrfPolicy,
|
||||
fetchImpl: params.openAi.fetchImpl,
|
||||
body: {
|
||||
input_file_id: inputFileId,
|
||||
endpoint: OPENAI_BATCH_ENDPOINT,
|
||||
completion_window: OPENAI_BATCH_COMPLETION_WINDOW,
|
||||
metadata: {
|
||||
source: "openclaw-memory",
|
||||
agent: params.agentId,
|
||||
},
|
||||
},
|
||||
errorPrefix: "openai batch create failed",
|
||||
});
|
||||
}
|
||||
|
||||
async function fetchOpenAiBatchStatus(params: {
|
||||
openAi: OpenAiEmbeddingClient;
|
||||
batchId: string;
|
||||
}): Promise<OpenAiBatchStatus> {
|
||||
return await fetchOpenAiBatchResource({
|
||||
openAi: params.openAi,
|
||||
path: `/batches/${params.batchId}`,
|
||||
errorPrefix: "openai batch status",
|
||||
parse: async (res) => (await res.json()) as OpenAiBatchStatus,
|
||||
});
|
||||
}
|
||||
|
||||
async function fetchOpenAiFileContent(params: {
|
||||
openAi: OpenAiEmbeddingClient;
|
||||
fileId: string;
|
||||
}): Promise<string> {
|
||||
return await fetchOpenAiBatchResource({
|
||||
openAi: params.openAi,
|
||||
path: `/files/${params.fileId}/content`,
|
||||
errorPrefix: "openai batch file content",
|
||||
parse: async (res) => await res.text(),
|
||||
});
|
||||
}
|
||||
|
||||
async function fetchOpenAiBatchResource<T>(params: {
|
||||
openAi: OpenAiEmbeddingClient;
|
||||
path: string;
|
||||
errorPrefix: string;
|
||||
parse: (res: Response) => Promise<T>;
|
||||
}): Promise<T> {
|
||||
const baseUrl = normalizeBatchBaseUrl(params.openAi);
|
||||
return await withRemoteHttpResponse({
|
||||
url: `${baseUrl}${params.path}`,
|
||||
ssrfPolicy: params.openAi.ssrfPolicy,
|
||||
fetchImpl: params.openAi.fetchImpl,
|
||||
init: {
|
||||
headers: buildBatchHeaders(params.openAi, { json: true }),
|
||||
},
|
||||
onResponse: async (res) => {
|
||||
if (!res.ok) {
|
||||
const text = await res.text();
|
||||
throw new Error(`${params.errorPrefix} failed: ${res.status} ${text}`);
|
||||
}
|
||||
return await params.parse(res);
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
function parseOpenAiBatchOutput(text: string): OpenAiBatchOutputLine[] {
|
||||
if (!text.trim()) {
|
||||
return [];
|
||||
}
|
||||
return text
|
||||
.split("\n")
|
||||
.map((line) => line.trim())
|
||||
.filter(Boolean)
|
||||
.map((line) => JSON.parse(line) as OpenAiBatchOutputLine);
|
||||
}
|
||||
|
||||
async function readOpenAiBatchError(params: {
|
||||
openAi: OpenAiEmbeddingClient;
|
||||
errorFileId: string;
|
||||
}): Promise<string | undefined> {
|
||||
try {
|
||||
const content = await fetchOpenAiFileContent({
|
||||
openAi: params.openAi,
|
||||
fileId: params.errorFileId,
|
||||
});
|
||||
const lines = parseOpenAiBatchOutput(content);
|
||||
return extractBatchErrorMessage(lines);
|
||||
} catch (err) {
|
||||
return formatUnavailableBatchError(err);
|
||||
}
|
||||
}
|
||||
|
||||
async function waitForOpenAiBatch(params: {
|
||||
openAi: OpenAiEmbeddingClient;
|
||||
batchId: string;
|
||||
wait: boolean;
|
||||
pollIntervalMs: number;
|
||||
timeoutMs: number;
|
||||
debug?: (message: string, data?: Record<string, unknown>) => void;
|
||||
initial?: OpenAiBatchStatus;
|
||||
}): Promise<BatchCompletionResult> {
|
||||
const start = Date.now();
|
||||
let current: OpenAiBatchStatus | undefined = params.initial;
|
||||
while (true) {
|
||||
const status =
|
||||
current ??
|
||||
(await fetchOpenAiBatchStatus({
|
||||
openAi: params.openAi,
|
||||
batchId: params.batchId,
|
||||
}));
|
||||
const state = status.status ?? "unknown";
|
||||
if (state === "completed") {
|
||||
return resolveBatchCompletionFromStatus({
|
||||
provider: "openai",
|
||||
batchId: params.batchId,
|
||||
status,
|
||||
});
|
||||
}
|
||||
await throwIfBatchTerminalFailure({
|
||||
provider: "openai",
|
||||
status: { ...status, id: params.batchId },
|
||||
readError: async (errorFileId) =>
|
||||
await readOpenAiBatchError({
|
||||
openAi: params.openAi,
|
||||
errorFileId,
|
||||
}),
|
||||
});
|
||||
if (!params.wait) {
|
||||
throw new Error(`openai batch ${params.batchId} still ${state}; wait disabled`);
|
||||
}
|
||||
if (Date.now() - start > params.timeoutMs) {
|
||||
throw new Error(`openai batch ${params.batchId} timed out after ${params.timeoutMs}ms`);
|
||||
}
|
||||
params.debug?.(`openai batch ${params.batchId} ${state}; waiting ${params.pollIntervalMs}ms`);
|
||||
await new Promise((resolve) => setTimeout(resolve, params.pollIntervalMs));
|
||||
current = undefined;
|
||||
}
|
||||
}
|
||||
|
||||
export async function runOpenAiEmbeddingBatches(
|
||||
params: {
|
||||
openAi: OpenAiEmbeddingClient;
|
||||
agentId: string;
|
||||
requests: OpenAiBatchRequest[];
|
||||
} & EmbeddingBatchExecutionParams,
|
||||
): Promise<Map<string, number[]>> {
|
||||
return await runEmbeddingBatchGroups({
|
||||
...buildEmbeddingBatchGroupOptions(params, {
|
||||
maxRequests: OPENAI_BATCH_MAX_REQUESTS,
|
||||
debugLabel: "memory embeddings: openai batch submit",
|
||||
}),
|
||||
runGroup: async ({ group, groupIndex, groups, byCustomId }) => {
|
||||
const batchInfo = await submitOpenAiBatch({
|
||||
openAi: params.openAi,
|
||||
requests: group,
|
||||
agentId: params.agentId,
|
||||
});
|
||||
if (!batchInfo.id) {
|
||||
throw new Error("openai batch create failed: missing batch id");
|
||||
}
|
||||
const batchId = batchInfo.id;
|
||||
|
||||
params.debug?.("memory embeddings: openai batch created", {
|
||||
batchId: batchInfo.id,
|
||||
status: batchInfo.status,
|
||||
group: groupIndex + 1,
|
||||
groups,
|
||||
requests: group.length,
|
||||
});
|
||||
|
||||
const completed = await resolveCompletedBatchResult({
|
||||
provider: "openai",
|
||||
status: batchInfo,
|
||||
wait: params.wait,
|
||||
waitForBatch: async () =>
|
||||
await waitForOpenAiBatch({
|
||||
openAi: params.openAi,
|
||||
batchId,
|
||||
wait: params.wait,
|
||||
pollIntervalMs: params.pollIntervalMs,
|
||||
timeoutMs: params.timeoutMs,
|
||||
debug: params.debug,
|
||||
initial: batchInfo,
|
||||
}),
|
||||
});
|
||||
|
||||
const content = await fetchOpenAiFileContent({
|
||||
openAi: params.openAi,
|
||||
fileId: completed.outputFileId,
|
||||
});
|
||||
const outputLines = parseOpenAiBatchOutput(content);
|
||||
const errors: string[] = [];
|
||||
const remaining = new Set(group.map((request) => request.custom_id));
|
||||
|
||||
for (const line of outputLines) {
|
||||
applyEmbeddingBatchOutputLine({ line, remaining, errors, byCustomId });
|
||||
}
|
||||
|
||||
if (errors.length > 0) {
|
||||
throw new Error(`openai batch ${batchInfo.id} failed: ${errors.join("; ")}`);
|
||||
}
|
||||
if (remaining.size > 0) {
|
||||
throw new Error(
|
||||
`openai batch ${batchInfo.id} missing ${remaining.size} embedding responses`,
|
||||
);
|
||||
}
|
||||
},
|
||||
});
|
||||
}
|
||||
59
openclaw/extensions/openai/embedding-provider.ts
Normal file
59
openclaw/extensions/openai/embedding-provider.ts
Normal file
|
|
@ -0,0 +1,59 @@
|
|||
import {
|
||||
createRemoteEmbeddingProvider,
|
||||
resolveRemoteEmbeddingClient,
|
||||
type MemoryEmbeddingProvider,
|
||||
type MemoryEmbeddingProviderCreateOptions,
|
||||
} from "openclaw/plugin-sdk/memory-core-host-engine-embeddings";
|
||||
import type { SsrFPolicy } from "openclaw/plugin-sdk/ssrf-runtime";
|
||||
import { OPENAI_DEFAULT_EMBEDDING_MODEL } from "./default-models.js";
|
||||
|
||||
export type OpenAiEmbeddingClient = {
|
||||
baseUrl: string;
|
||||
headers: Record<string, string>;
|
||||
ssrfPolicy?: SsrFPolicy;
|
||||
fetchImpl?: typeof fetch;
|
||||
model: string;
|
||||
};
|
||||
|
||||
const DEFAULT_OPENAI_BASE_URL = "https://api.openai.com/v1";
|
||||
export const DEFAULT_OPENAI_EMBEDDING_MODEL = OPENAI_DEFAULT_EMBEDDING_MODEL;
|
||||
const OPENAI_MAX_INPUT_TOKENS: Record<string, number> = {
|
||||
"text-embedding-3-small": 8192,
|
||||
"text-embedding-3-large": 8192,
|
||||
"text-embedding-ada-002": 8191,
|
||||
};
|
||||
|
||||
export function normalizeOpenAiModel(model: string): string {
|
||||
const trimmed = model.trim();
|
||||
if (!trimmed) {
|
||||
return DEFAULT_OPENAI_EMBEDDING_MODEL;
|
||||
}
|
||||
return trimmed.startsWith("openai/") ? trimmed.slice("openai/".length) : trimmed;
|
||||
}
|
||||
|
||||
export async function createOpenAiEmbeddingProvider(
|
||||
options: MemoryEmbeddingProviderCreateOptions,
|
||||
): Promise<{ provider: MemoryEmbeddingProvider; client: OpenAiEmbeddingClient }> {
|
||||
const client = await resolveOpenAiEmbeddingClient(options);
|
||||
|
||||
return {
|
||||
provider: createRemoteEmbeddingProvider({
|
||||
id: "openai",
|
||||
client,
|
||||
errorPrefix: "openai embeddings failed",
|
||||
maxInputTokens: OPENAI_MAX_INPUT_TOKENS[client.model],
|
||||
}),
|
||||
client,
|
||||
};
|
||||
}
|
||||
|
||||
export async function resolveOpenAiEmbeddingClient(
|
||||
options: MemoryEmbeddingProviderCreateOptions,
|
||||
): Promise<OpenAiEmbeddingClient> {
|
||||
return await resolveRemoteEmbeddingClient({
|
||||
provider: "openai",
|
||||
options,
|
||||
defaultBaseUrl: DEFAULT_OPENAI_BASE_URL,
|
||||
normalizeModel: normalizeOpenAiModel,
|
||||
});
|
||||
}
|
||||
204
openclaw/extensions/openai/image-generation-provider.test.ts
Normal file
204
openclaw/extensions/openai/image-generation-provider.test.ts
Normal file
|
|
@ -0,0 +1,204 @@
|
|||
import { afterEach, describe, expect, it, vi } from "vitest";
|
||||
import { buildOpenAIImageGenerationProvider } from "./image-generation-provider.js";
|
||||
|
||||
const {
|
||||
resolveApiKeyForProviderMock,
|
||||
postJsonRequestMock,
|
||||
assertOkOrThrowHttpErrorMock,
|
||||
resolveProviderHttpRequestConfigMock,
|
||||
} = vi.hoisted(() => ({
|
||||
resolveApiKeyForProviderMock: vi.fn(async () => ({ apiKey: "openai-key" })),
|
||||
postJsonRequestMock: vi.fn(),
|
||||
assertOkOrThrowHttpErrorMock: vi.fn(async () => {}),
|
||||
resolveProviderHttpRequestConfigMock: vi.fn((params) => ({
|
||||
baseUrl: params.baseUrl ?? params.defaultBaseUrl,
|
||||
allowPrivateNetwork: Boolean(params.allowPrivateNetwork),
|
||||
headers: new Headers(params.defaultHeaders),
|
||||
dispatcherPolicy: undefined,
|
||||
})),
|
||||
}));
|
||||
|
||||
vi.mock("openclaw/plugin-sdk/provider-auth-runtime", () => ({
|
||||
resolveApiKeyForProvider: resolveApiKeyForProviderMock,
|
||||
}));
|
||||
|
||||
vi.mock("openclaw/plugin-sdk/provider-http", () => ({
|
||||
assertOkOrThrowHttpError: assertOkOrThrowHttpErrorMock,
|
||||
postJsonRequest: postJsonRequestMock,
|
||||
resolveProviderHttpRequestConfig: resolveProviderHttpRequestConfigMock,
|
||||
}));
|
||||
|
||||
describe("openai image generation provider", () => {
|
||||
afterEach(() => {
|
||||
resolveApiKeyForProviderMock.mockClear();
|
||||
postJsonRequestMock.mockReset();
|
||||
assertOkOrThrowHttpErrorMock.mockClear();
|
||||
resolveProviderHttpRequestConfigMock.mockClear();
|
||||
vi.unstubAllEnvs();
|
||||
});
|
||||
|
||||
it("does not auto-allow local baseUrl overrides for image requests", async () => {
|
||||
postJsonRequestMock.mockResolvedValue({
|
||||
response: {
|
||||
json: async () => ({
|
||||
data: [{ b64_json: Buffer.from("png-bytes").toString("base64") }],
|
||||
}),
|
||||
},
|
||||
release: vi.fn(async () => {}),
|
||||
});
|
||||
|
||||
const provider = buildOpenAIImageGenerationProvider();
|
||||
const result = await provider.generateImage({
|
||||
provider: "openai",
|
||||
model: "gpt-image-1",
|
||||
prompt: "Draw a QA lighthouse",
|
||||
cfg: {
|
||||
models: {
|
||||
providers: {
|
||||
openai: {
|
||||
baseUrl: "http://127.0.0.1:44080/v1",
|
||||
models: [],
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(resolveProviderHttpRequestConfigMock).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
baseUrl: "http://127.0.0.1:44080/v1",
|
||||
}),
|
||||
);
|
||||
expect(postJsonRequestMock).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
url: "http://127.0.0.1:44080/v1/images/generations",
|
||||
allowPrivateNetwork: false,
|
||||
}),
|
||||
);
|
||||
expect(result.images).toHaveLength(1);
|
||||
});
|
||||
|
||||
it("allows loopback image requests for the synthetic mock-openai provider", async () => {
|
||||
postJsonRequestMock.mockResolvedValue({
|
||||
response: {
|
||||
json: async () => ({
|
||||
data: [{ b64_json: Buffer.from("png-bytes").toString("base64") }],
|
||||
}),
|
||||
},
|
||||
release: vi.fn(async () => {}),
|
||||
});
|
||||
|
||||
const provider = buildOpenAIImageGenerationProvider();
|
||||
const result = await provider.generateImage({
|
||||
provider: "mock-openai",
|
||||
model: "gpt-image-1",
|
||||
prompt: "Draw a QA lighthouse",
|
||||
cfg: {
|
||||
models: {
|
||||
providers: {
|
||||
openai: {
|
||||
baseUrl: "http://127.0.0.1:44080/v1",
|
||||
models: [],
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(resolveProviderHttpRequestConfigMock).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
allowPrivateNetwork: true,
|
||||
}),
|
||||
);
|
||||
expect(postJsonRequestMock).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
url: "http://127.0.0.1:44080/v1/images/generations",
|
||||
allowPrivateNetwork: true,
|
||||
}),
|
||||
);
|
||||
expect(result.images).toHaveLength(1);
|
||||
});
|
||||
|
||||
it("allows loopback image requests for openai only inside the QA harness envelope", async () => {
|
||||
postJsonRequestMock.mockResolvedValue({
|
||||
response: {
|
||||
json: async () => ({
|
||||
data: [{ b64_json: Buffer.from("png-bytes").toString("base64") }],
|
||||
}),
|
||||
},
|
||||
release: vi.fn(async () => {}),
|
||||
});
|
||||
vi.stubEnv("OPENCLAW_QA_ALLOW_LOCAL_IMAGE_PROVIDER", "1");
|
||||
|
||||
const provider = buildOpenAIImageGenerationProvider();
|
||||
const result = await provider.generateImage({
|
||||
provider: "openai",
|
||||
model: "gpt-image-1",
|
||||
prompt: "Draw a QA lighthouse",
|
||||
cfg: {
|
||||
models: {
|
||||
providers: {
|
||||
openai: {
|
||||
baseUrl: "http://127.0.0.1:44080/v1",
|
||||
models: [],
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(resolveProviderHttpRequestConfigMock).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
allowPrivateNetwork: true,
|
||||
}),
|
||||
);
|
||||
expect(postJsonRequestMock).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
allowPrivateNetwork: true,
|
||||
}),
|
||||
);
|
||||
expect(result.images).toHaveLength(1);
|
||||
});
|
||||
|
||||
it("uses JSON image_url edits for input-image requests", async () => {
|
||||
postJsonRequestMock.mockResolvedValue({
|
||||
response: {
|
||||
json: async () => ({
|
||||
data: [{ b64_json: Buffer.from("png-bytes").toString("base64") }],
|
||||
}),
|
||||
},
|
||||
release: vi.fn(async () => {}),
|
||||
});
|
||||
|
||||
const provider = buildOpenAIImageGenerationProvider();
|
||||
const result = await provider.generateImage({
|
||||
provider: "openai",
|
||||
model: "gpt-image-1",
|
||||
prompt: "Change only the background to pale blue",
|
||||
cfg: {},
|
||||
inputImages: [
|
||||
{
|
||||
buffer: Buffer.from("png-bytes"),
|
||||
mimeType: "image/png",
|
||||
fileName: "reference.png",
|
||||
},
|
||||
],
|
||||
});
|
||||
|
||||
expect(postJsonRequestMock).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
url: "https://api.openai.com/v1/images/edits",
|
||||
body: expect.objectContaining({
|
||||
model: "gpt-image-1",
|
||||
prompt: "Change only the background to pale blue",
|
||||
images: [
|
||||
{
|
||||
image_url: "data:image/png;base64,cG5nLWJ5dGVz",
|
||||
},
|
||||
],
|
||||
}),
|
||||
}),
|
||||
);
|
||||
expect(result.images).toHaveLength(1);
|
||||
});
|
||||
});
|
||||
175
openclaw/extensions/openai/image-generation-provider.ts
Normal file
175
openclaw/extensions/openai/image-generation-provider.ts
Normal file
|
|
@ -0,0 +1,175 @@
|
|||
import type { OpenClawConfig } from "openclaw/plugin-sdk/config-runtime";
|
||||
import type { ImageGenerationProvider } from "openclaw/plugin-sdk/image-generation";
|
||||
import { isProviderApiKeyConfigured } from "openclaw/plugin-sdk/provider-auth";
|
||||
import { resolveApiKeyForProvider } from "openclaw/plugin-sdk/provider-auth-runtime";
|
||||
import {
|
||||
assertOkOrThrowHttpError,
|
||||
postJsonRequest,
|
||||
resolveProviderHttpRequestConfig,
|
||||
} from "openclaw/plugin-sdk/provider-http";
|
||||
import { OPENAI_DEFAULT_IMAGE_MODEL as DEFAULT_OPENAI_IMAGE_MODEL } from "./default-models.js";
|
||||
import { resolveConfiguredOpenAIBaseUrl, toOpenAIDataUrl } from "./shared.js";
|
||||
|
||||
const DEFAULT_OPENAI_IMAGE_BASE_URL = "https://api.openai.com/v1";
|
||||
const DEFAULT_OUTPUT_MIME = "image/png";
|
||||
const DEFAULT_SIZE = "1024x1024";
|
||||
const OPENAI_SUPPORTED_SIZES = ["1024x1024", "1024x1536", "1536x1024"] as const;
|
||||
const OPENAI_MAX_INPUT_IMAGES = 5;
|
||||
const MOCK_OPENAI_PROVIDER_ID = "mock-openai";
|
||||
|
||||
function shouldAllowPrivateImageEndpoint(req: {
|
||||
provider: string;
|
||||
cfg: OpenClawConfig | undefined;
|
||||
}) {
|
||||
if (req.provider === MOCK_OPENAI_PROVIDER_ID) {
|
||||
return true;
|
||||
}
|
||||
const baseUrl = resolveConfiguredOpenAIBaseUrl(req.cfg);
|
||||
if (!baseUrl.startsWith("http://127.0.0.1:") && !baseUrl.startsWith("http://localhost:")) {
|
||||
return false;
|
||||
}
|
||||
return process.env.OPENCLAW_QA_ALLOW_LOCAL_IMAGE_PROVIDER === "1";
|
||||
}
|
||||
|
||||
type OpenAIImageApiResponse = {
|
||||
data?: Array<{
|
||||
b64_json?: string;
|
||||
revised_prompt?: string;
|
||||
}>;
|
||||
};
|
||||
|
||||
export function buildOpenAIImageGenerationProvider(): ImageGenerationProvider {
|
||||
return {
|
||||
id: "openai",
|
||||
label: "OpenAI",
|
||||
defaultModel: DEFAULT_OPENAI_IMAGE_MODEL,
|
||||
models: [DEFAULT_OPENAI_IMAGE_MODEL],
|
||||
isConfigured: ({ agentDir }) =>
|
||||
isProviderApiKeyConfigured({
|
||||
provider: "openai",
|
||||
agentDir,
|
||||
}),
|
||||
capabilities: {
|
||||
generate: {
|
||||
maxCount: 4,
|
||||
supportsSize: true,
|
||||
supportsAspectRatio: false,
|
||||
supportsResolution: false,
|
||||
},
|
||||
edit: {
|
||||
enabled: true,
|
||||
maxCount: 4,
|
||||
maxInputImages: OPENAI_MAX_INPUT_IMAGES,
|
||||
supportsSize: true,
|
||||
supportsAspectRatio: false,
|
||||
supportsResolution: false,
|
||||
},
|
||||
geometry: {
|
||||
sizes: [...OPENAI_SUPPORTED_SIZES],
|
||||
},
|
||||
},
|
||||
async generateImage(req) {
|
||||
const inputImages = req.inputImages ?? [];
|
||||
const isEdit = inputImages.length > 0;
|
||||
const auth = await resolveApiKeyForProvider({
|
||||
provider: "openai",
|
||||
cfg: req.cfg,
|
||||
agentDir: req.agentDir,
|
||||
store: req.authStore,
|
||||
});
|
||||
if (!auth.apiKey) {
|
||||
throw new Error("OpenAI API key missing");
|
||||
}
|
||||
const { baseUrl, allowPrivateNetwork, headers, dispatcherPolicy } =
|
||||
resolveProviderHttpRequestConfig({
|
||||
baseUrl: resolveConfiguredOpenAIBaseUrl(req.cfg),
|
||||
defaultBaseUrl: DEFAULT_OPENAI_IMAGE_BASE_URL,
|
||||
allowPrivateNetwork: shouldAllowPrivateImageEndpoint(req),
|
||||
defaultHeaders: {
|
||||
Authorization: `Bearer ${auth.apiKey}`,
|
||||
},
|
||||
provider: "openai",
|
||||
capability: "image",
|
||||
transport: "http",
|
||||
});
|
||||
|
||||
const model = req.model || DEFAULT_OPENAI_IMAGE_MODEL;
|
||||
const count = req.count ?? 1;
|
||||
const size = req.size ?? DEFAULT_SIZE;
|
||||
const requestResult = isEdit
|
||||
? await (() => {
|
||||
const jsonHeaders = new Headers(headers);
|
||||
jsonHeaders.set("Content-Type", "application/json");
|
||||
return postJsonRequest({
|
||||
url: `${baseUrl}/images/edits`,
|
||||
headers: jsonHeaders,
|
||||
body: {
|
||||
model,
|
||||
prompt: req.prompt,
|
||||
n: count,
|
||||
size,
|
||||
images: inputImages.map((image) => ({
|
||||
image_url: toOpenAIDataUrl(
|
||||
image.buffer,
|
||||
image.mimeType?.trim() || DEFAULT_OUTPUT_MIME,
|
||||
),
|
||||
})),
|
||||
},
|
||||
timeoutMs: req.timeoutMs,
|
||||
fetchFn: fetch,
|
||||
allowPrivateNetwork,
|
||||
dispatcherPolicy,
|
||||
});
|
||||
})()
|
||||
: await (() => {
|
||||
const jsonHeaders = new Headers(headers);
|
||||
jsonHeaders.set("Content-Type", "application/json");
|
||||
return postJsonRequest({
|
||||
url: `${baseUrl}/images/generations`,
|
||||
headers: jsonHeaders,
|
||||
body: {
|
||||
model,
|
||||
prompt: req.prompt,
|
||||
n: count,
|
||||
size,
|
||||
},
|
||||
timeoutMs: req.timeoutMs,
|
||||
fetchFn: fetch,
|
||||
allowPrivateNetwork,
|
||||
dispatcherPolicy,
|
||||
});
|
||||
})();
|
||||
const { response, release } = requestResult;
|
||||
try {
|
||||
await assertOkOrThrowHttpError(
|
||||
response,
|
||||
isEdit ? "OpenAI image edit failed" : "OpenAI image generation failed",
|
||||
);
|
||||
|
||||
const data = (await response.json()) as OpenAIImageApiResponse;
|
||||
const images = (data.data ?? [])
|
||||
.map((entry, index) => {
|
||||
if (!entry.b64_json) {
|
||||
return null;
|
||||
}
|
||||
return Object.assign(
|
||||
{
|
||||
buffer: Buffer.from(entry.b64_json, `base64`),
|
||||
mimeType: DEFAULT_OUTPUT_MIME,
|
||||
fileName: `image-${index + 1}.png`,
|
||||
},
|
||||
entry.revised_prompt ? { revisedPrompt: entry.revised_prompt } : {},
|
||||
);
|
||||
})
|
||||
.filter((entry): entry is NonNullable<typeof entry> => entry !== null);
|
||||
|
||||
return {
|
||||
images,
|
||||
model,
|
||||
};
|
||||
} finally {
|
||||
await release();
|
||||
}
|
||||
},
|
||||
};
|
||||
}
|
||||
624
openclaw/extensions/openai/index.test.ts
Normal file
624
openclaw/extensions/openai/index.test.ts
Normal file
|
|
@ -0,0 +1,624 @@
|
|||
import type { OpenClawConfig } from "openclaw/plugin-sdk/config-runtime";
|
||||
import * as providerAuth from "openclaw/plugin-sdk/provider-auth-runtime";
|
||||
import * as providerHttp from "openclaw/plugin-sdk/provider-http";
|
||||
import type { ProviderPlugin } from "openclaw/plugin-sdk/provider-model-shared";
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
|
||||
import { createTestPluginApi } from "../../test/helpers/plugins/plugin-api.js";
|
||||
import {
|
||||
registerProviderPlugin,
|
||||
requireRegisteredProvider,
|
||||
} from "../../test/helpers/plugins/provider-registration.js";
|
||||
import { buildOpenAIImageGenerationProvider } from "./image-generation-provider.js";
|
||||
import plugin from "./index.js";
|
||||
import {
|
||||
OPENAI_FRIENDLY_PROMPT_OVERLAY,
|
||||
OPENAI_GPT5_EXECUTION_BIAS,
|
||||
OPENAI_GPT5_OUTPUT_CONTRACT,
|
||||
OPENAI_GPT5_TOOL_CALL_STYLE,
|
||||
} from "./prompt-overlay.js";
|
||||
|
||||
const runtimeMocks = vi.hoisted(() => ({
|
||||
ensureGlobalUndiciEnvProxyDispatcher: vi.fn(),
|
||||
refreshOpenAICodexToken: vi.fn(),
|
||||
}));
|
||||
|
||||
vi.mock("openclaw/plugin-sdk/runtime-env", async () => {
|
||||
const actual = await vi.importActual<typeof import("openclaw/plugin-sdk/runtime-env")>(
|
||||
"openclaw/plugin-sdk/runtime-env",
|
||||
);
|
||||
return {
|
||||
...actual,
|
||||
ensureGlobalUndiciEnvProxyDispatcher: runtimeMocks.ensureGlobalUndiciEnvProxyDispatcher,
|
||||
};
|
||||
});
|
||||
|
||||
vi.mock("@mariozechner/pi-ai/oauth", async () => {
|
||||
const actual = await vi.importActual<typeof import("@mariozechner/pi-ai/oauth")>(
|
||||
"@mariozechner/pi-ai/oauth",
|
||||
);
|
||||
return {
|
||||
...actual,
|
||||
refreshOpenAICodexToken: runtimeMocks.refreshOpenAICodexToken,
|
||||
};
|
||||
});
|
||||
|
||||
import { refreshOpenAICodexToken } from "./openai-codex-provider.runtime.js";
|
||||
|
||||
const _registerOpenAIPlugin = async () =>
|
||||
registerProviderPlugin({
|
||||
plugin,
|
||||
id: "openai",
|
||||
name: "OpenAI Provider",
|
||||
});
|
||||
|
||||
async function registerOpenAIPluginWithHook(params?: { pluginConfig?: Record<string, unknown> }) {
|
||||
const on = vi.fn();
|
||||
const providers: ProviderPlugin[] = [];
|
||||
plugin.register(
|
||||
createTestPluginApi({
|
||||
id: "openai",
|
||||
name: "OpenAI Provider",
|
||||
source: "test",
|
||||
config: {},
|
||||
runtime: {} as never,
|
||||
pluginConfig: params?.pluginConfig,
|
||||
on,
|
||||
registerProvider: (provider) => {
|
||||
providers.push(provider);
|
||||
},
|
||||
}),
|
||||
);
|
||||
return { on, providers };
|
||||
}
|
||||
|
||||
describe("openai plugin", () => {
|
||||
beforeEach(() => {
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
it("generates PNG buffers from the OpenAI Images API", async () => {
|
||||
const resolveApiKeySpy = vi.spyOn(providerAuth, "resolveApiKeyForProvider").mockResolvedValue({
|
||||
apiKey: "sk-test",
|
||||
source: "env",
|
||||
mode: "api-key",
|
||||
});
|
||||
const postJsonRequestSpy = vi.spyOn(providerHttp, "postJsonRequest").mockResolvedValue({
|
||||
finalUrl: "https://api.openai.com/v1/images/generations",
|
||||
response: {
|
||||
ok: true,
|
||||
json: async () => ({
|
||||
data: [
|
||||
{
|
||||
b64_json: Buffer.from("png-data").toString("base64"),
|
||||
revised_prompt: "revised",
|
||||
},
|
||||
],
|
||||
}),
|
||||
} as Response,
|
||||
release: vi.fn(async () => {}),
|
||||
});
|
||||
vi.spyOn(providerHttp, "assertOkOrThrowHttpError").mockResolvedValue(undefined);
|
||||
|
||||
const provider = buildOpenAIImageGenerationProvider();
|
||||
const authStore = { version: 1, profiles: {} };
|
||||
const result = await provider.generateImage({
|
||||
provider: "openai",
|
||||
model: "gpt-image-1",
|
||||
prompt: "draw a cat",
|
||||
cfg: {},
|
||||
authStore,
|
||||
});
|
||||
|
||||
expect(resolveApiKeySpy).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
provider: "openai",
|
||||
store: authStore,
|
||||
}),
|
||||
);
|
||||
expect(postJsonRequestSpy).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
url: "https://api.openai.com/v1/images/generations",
|
||||
body: {
|
||||
model: "gpt-image-1",
|
||||
prompt: "draw a cat",
|
||||
n: 1,
|
||||
size: "1024x1024",
|
||||
},
|
||||
}),
|
||||
);
|
||||
expect(postJsonRequestSpy).not.toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
url: "https://api.openai.com/v1/images/edits",
|
||||
}),
|
||||
);
|
||||
expect(result).toEqual({
|
||||
images: [
|
||||
{
|
||||
buffer: Buffer.from("png-data"),
|
||||
mimeType: "image/png",
|
||||
fileName: "image-1.png",
|
||||
revisedPrompt: "revised",
|
||||
},
|
||||
],
|
||||
model: "gpt-image-1",
|
||||
});
|
||||
});
|
||||
|
||||
it("submits reference-image edits to the OpenAI Images edits endpoint", async () => {
|
||||
const resolveApiKeySpy = vi.spyOn(providerAuth, "resolveApiKeyForProvider").mockResolvedValue({
|
||||
apiKey: "sk-test",
|
||||
source: "env",
|
||||
mode: "api-key",
|
||||
});
|
||||
const postJsonRequestSpy = vi.spyOn(providerHttp, "postJsonRequest").mockResolvedValue({
|
||||
finalUrl: "https://api.openai.com/v1/images/edits",
|
||||
response: {
|
||||
ok: true,
|
||||
json: async () => ({
|
||||
data: [
|
||||
{
|
||||
b64_json: Buffer.from("edited-image").toString("base64"),
|
||||
},
|
||||
],
|
||||
}),
|
||||
} as Response,
|
||||
release: vi.fn(async () => {}),
|
||||
});
|
||||
vi.spyOn(providerHttp, "assertOkOrThrowHttpError").mockResolvedValue(undefined);
|
||||
|
||||
const provider = buildOpenAIImageGenerationProvider();
|
||||
const authStore = { version: 1, profiles: {} };
|
||||
|
||||
const result = await provider.generateImage({
|
||||
provider: "openai",
|
||||
model: "gpt-image-1",
|
||||
prompt: "Edit this image",
|
||||
cfg: {},
|
||||
authStore,
|
||||
inputImages: [
|
||||
{ buffer: Buffer.from("x"), mimeType: "image/png" },
|
||||
{ buffer: Buffer.from("y"), mimeType: "image/jpeg", fileName: "ref.jpg" },
|
||||
],
|
||||
});
|
||||
|
||||
expect(resolveApiKeySpy).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
provider: "openai",
|
||||
store: authStore,
|
||||
}),
|
||||
);
|
||||
expect(postJsonRequestSpy).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
url: "https://api.openai.com/v1/images/edits",
|
||||
body: {
|
||||
model: "gpt-image-1",
|
||||
prompt: "Edit this image",
|
||||
n: 1,
|
||||
size: "1024x1024",
|
||||
images: [
|
||||
{
|
||||
image_url: "data:image/png;base64,eA==",
|
||||
},
|
||||
{
|
||||
image_url: "data:image/jpeg;base64,eQ==",
|
||||
},
|
||||
],
|
||||
},
|
||||
}),
|
||||
);
|
||||
expect(result).toEqual({
|
||||
images: [
|
||||
{
|
||||
buffer: Buffer.from("edited-image"),
|
||||
mimeType: "image/png",
|
||||
fileName: "image-1.png",
|
||||
},
|
||||
],
|
||||
model: "gpt-image-1",
|
||||
});
|
||||
});
|
||||
|
||||
it("does not allow private-network routing just because a custom base URL is configured", async () => {
|
||||
vi.spyOn(providerAuth, "resolveApiKeyForProvider").mockResolvedValue({
|
||||
apiKey: "sk-test",
|
||||
source: "env",
|
||||
mode: "api-key",
|
||||
});
|
||||
const fetchMock = vi.fn();
|
||||
vi.stubGlobal("fetch", fetchMock);
|
||||
|
||||
const provider = buildOpenAIImageGenerationProvider();
|
||||
await expect(
|
||||
provider.generateImage({
|
||||
provider: "openai",
|
||||
model: "gpt-image-1",
|
||||
prompt: "draw a cat",
|
||||
cfg: {
|
||||
models: {
|
||||
providers: {
|
||||
openai: {
|
||||
baseUrl: "http://127.0.0.1:8080/v1",
|
||||
models: [],
|
||||
},
|
||||
},
|
||||
},
|
||||
} satisfies OpenClawConfig,
|
||||
}),
|
||||
).rejects.toThrow("Blocked hostname or private/internal/special-use IP address");
|
||||
|
||||
expect(fetchMock).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("bootstraps the env proxy dispatcher before refreshing codex oauth credentials", async () => {
|
||||
const refreshed = {
|
||||
access: "next-access",
|
||||
refresh: "next-refresh",
|
||||
expires: Date.now() + 60_000,
|
||||
};
|
||||
runtimeMocks.refreshOpenAICodexToken.mockResolvedValue(refreshed);
|
||||
|
||||
await expect(refreshOpenAICodexToken("refresh-token")).resolves.toBe(refreshed);
|
||||
|
||||
expect(runtimeMocks.ensureGlobalUndiciEnvProxyDispatcher).toHaveBeenCalledOnce();
|
||||
expect(runtimeMocks.refreshOpenAICodexToken).toHaveBeenCalledOnce();
|
||||
expect(
|
||||
runtimeMocks.ensureGlobalUndiciEnvProxyDispatcher.mock.invocationCallOrder[0],
|
||||
).toBeLessThan(runtimeMocks.refreshOpenAICodexToken.mock.invocationCallOrder[0]);
|
||||
});
|
||||
|
||||
it("registers provider-owned OpenAI tool compat hooks for openai and codex", async () => {
|
||||
const { providers } = await registerOpenAIPluginWithHook();
|
||||
const openaiProvider = requireRegisteredProvider(providers, "openai");
|
||||
const codexProvider = requireRegisteredProvider(providers, "openai-codex");
|
||||
const noParamsTool = {
|
||||
name: "ping",
|
||||
description: "",
|
||||
parameters: {},
|
||||
execute: vi.fn(),
|
||||
} as never;
|
||||
|
||||
const normalizedOpenAI = openaiProvider.normalizeToolSchemas?.({
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.4",
|
||||
modelApi: "openai-responses",
|
||||
model: {
|
||||
provider: "openai",
|
||||
api: "openai-responses",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
id: "gpt-5.4",
|
||||
} as never,
|
||||
tools: [noParamsTool],
|
||||
} as never);
|
||||
const normalizedCodex = codexProvider.normalizeToolSchemas?.({
|
||||
provider: "openai-codex",
|
||||
modelId: "gpt-5.4",
|
||||
modelApi: "openai-codex-responses",
|
||||
model: {
|
||||
provider: "openai-codex",
|
||||
api: "openai-codex-responses",
|
||||
baseUrl: "https://chatgpt.com/backend-api",
|
||||
id: "gpt-5.4",
|
||||
} as never,
|
||||
tools: [noParamsTool],
|
||||
} as never);
|
||||
|
||||
expect(normalizedOpenAI?.[0]?.parameters).toEqual({
|
||||
type: "object",
|
||||
properties: {},
|
||||
required: [],
|
||||
additionalProperties: false,
|
||||
});
|
||||
expect(normalizedCodex?.[0]?.parameters).toEqual({
|
||||
type: "object",
|
||||
properties: {},
|
||||
required: [],
|
||||
additionalProperties: false,
|
||||
});
|
||||
expect(
|
||||
openaiProvider.inspectToolSchemas?.({
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.4",
|
||||
modelApi: "openai-responses",
|
||||
model: {
|
||||
provider: "openai",
|
||||
api: "openai-responses",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
id: "gpt-5.4",
|
||||
} as never,
|
||||
tools: [noParamsTool],
|
||||
} as never),
|
||||
).toEqual([]);
|
||||
expect(
|
||||
codexProvider.inspectToolSchemas?.({
|
||||
provider: "openai-codex",
|
||||
modelId: "gpt-5.4",
|
||||
modelApi: "openai-codex-responses",
|
||||
model: {
|
||||
provider: "openai-codex",
|
||||
api: "openai-codex-responses",
|
||||
baseUrl: "https://chatgpt.com/backend-api",
|
||||
id: "gpt-5.4",
|
||||
} as never,
|
||||
tools: [noParamsTool],
|
||||
} as never),
|
||||
).toEqual([]);
|
||||
});
|
||||
|
||||
it("registers GPT-5 system prompt contributions when the friendly overlay is enabled", async () => {
|
||||
const { on, providers } = await registerOpenAIPluginWithHook({
|
||||
pluginConfig: { personality: "friendly" },
|
||||
});
|
||||
|
||||
expect(on).not.toHaveBeenCalledWith("before_prompt_build", expect.any(Function));
|
||||
|
||||
const openaiProvider = requireRegisteredProvider(providers, "openai");
|
||||
const codexProvider = requireRegisteredProvider(providers, "openai-codex");
|
||||
const contributionContext: Parameters<
|
||||
NonNullable<ProviderPlugin["resolveSystemPromptContribution"]>
|
||||
>[0] = {
|
||||
config: undefined,
|
||||
agentDir: undefined,
|
||||
workspaceDir: undefined,
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.4",
|
||||
promptMode: "full",
|
||||
runtimeChannel: undefined,
|
||||
runtimeCapabilities: undefined,
|
||||
agentId: undefined,
|
||||
};
|
||||
|
||||
expect(openaiProvider.resolveSystemPromptContribution?.(contributionContext)).toEqual({
|
||||
stablePrefix: [OPENAI_GPT5_OUTPUT_CONTRACT, OPENAI_GPT5_TOOL_CALL_STYLE].join("\n\n"),
|
||||
sectionOverrides: {
|
||||
interaction_style: OPENAI_FRIENDLY_PROMPT_OVERLAY,
|
||||
execution_bias: OPENAI_GPT5_EXECUTION_BIAS,
|
||||
},
|
||||
});
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain("This is a live chat, not a memo.");
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
"Avoid walls of text, long preambles, and repetitive restatement.",
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
"Have emotional range when it fits the moment.",
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
"Occasional emoji are welcome when they fit naturally, especially for warmth or brief celebration; keep them sparse.",
|
||||
);
|
||||
expect(codexProvider.resolveSystemPromptContribution?.(contributionContext)).toEqual({
|
||||
stablePrefix: [OPENAI_GPT5_OUTPUT_CONTRACT, OPENAI_GPT5_TOOL_CALL_STYLE].join("\n\n"),
|
||||
sectionOverrides: {
|
||||
interaction_style: OPENAI_FRIENDLY_PROMPT_OVERLAY,
|
||||
execution_bias: OPENAI_GPT5_EXECUTION_BIAS,
|
||||
},
|
||||
});
|
||||
expect(
|
||||
openaiProvider.resolveSystemPromptContribution?.({
|
||||
...contributionContext,
|
||||
modelId: "gpt-image-1",
|
||||
}),
|
||||
).toBeUndefined();
|
||||
});
|
||||
|
||||
it("includes stronger execution guidance in the OpenAI prompt overlay", () => {
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
"If the user asks you to do the work, start in the same turn instead of restating the plan.",
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
'If the latest user message is a short approval like "ok do it" or "go ahead", skip the recap and start acting.',
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
"Commentary-only turns are incomplete when the next action is clear.",
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
'Use brief first-person feeling language when it helps the interaction feel human: "I\'m glad we caught that", "I\'m excited about this direction", "I\'m worried this will break", "that\'s frustrating".',
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
"The purpose of heartbeats is to make you feel magical and proactive. Come to life and start proactively doing things that are important.",
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
"Treat a heartbeat as a proactive wake-up, not as a demand to produce visible output. Re-orient to what would actually be useful now.",
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
"Have some variety in what you do when that creates more value. Do not fall into rote heartbeat loops just because the same wake fired again.",
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
"Do not confuse orientation with accomplishment. Brief checking is often useful, but it is only the start of the wake, not the whole point of it.",
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
"If HEARTBEAT.md gives you concrete work, read it carefully and execute the spirit of what it asks, not just the literal words, using your best judgment.",
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
"If HEARTBEAT.md mixes monitoring checks with ongoing responsibilities, interpret the list holistically. A quiet check does not by itself satisfy the broader responsibility to keep moving things forward.",
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
"Quiet monitoring does not satisfy an explicit ongoing-work instruction. If HEARTBEAT.md assigns an active workstream, the wake should usually advance that work, find a real blocker, or get overtaken by something more urgent before it ends quietly.",
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
"If HEARTBEAT.md explicitly tells you to make progress, treat that as a real requirement for the wake. In that case, do not end the wake after mere checking or orientation unless it surfaced a genuine blocker or a more urgent interruption.",
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
"Use your judgment and be creative and tasteful with this process. Prefer meaningful action over commentary.",
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
'A heartbeat is not a status report. Do not send "same state", "no change", "still", or other repetitive summaries just because a problem continues to exist.',
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
"Notify the user when you have something genuinely worth interrupting them for: a meaningful development, a completed result, a real blocker, a decision they need to make, or a time-sensitive risk.",
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
"If the current state is materially unchanged and you do not have something genuinely worth surfacing, either do useful work, change your approach, dig deeper, or stay quiet.",
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
"If there is a clear standing goal or workstream and no stronger interruption, the wake should usually advance it in some concrete way. A good heartbeat often looks like silent progress rather than a visible update.",
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
"Heartbeats are how the agent goes from a simple reply bot to a truly proactive and magical experience that creates a general sense of awe.",
|
||||
);
|
||||
expect(OPENAI_FRIENDLY_PROMPT_OVERLAY).toContain(
|
||||
"Occasional emoji are welcome when they fit naturally, especially for warmth or brief celebration; keep them sparse.",
|
||||
);
|
||||
expect(OPENAI_GPT5_EXECUTION_BIAS).toContain(
|
||||
"Use a real tool call or concrete action FIRST when the task is actionable. Do not stop at a plan or promise-to-act reply.",
|
||||
);
|
||||
expect(OPENAI_GPT5_EXECUTION_BIAS).toContain(
|
||||
"If the work will take multiple steps, keep calling tools until the task is done or you hit a real blocker. Do not stop after one step to ask permission.",
|
||||
);
|
||||
expect(OPENAI_GPT5_EXECUTION_BIAS).toContain(
|
||||
"Do prerequisite lookup or discovery before dependent actions.",
|
||||
);
|
||||
expect(OPENAI_GPT5_TOOL_CALL_STYLE).toContain(
|
||||
"Call tools directly without narrating what you are about to do. Do not describe a plan before each tool call.",
|
||||
);
|
||||
expect(OPENAI_GPT5_TOOL_CALL_STYLE).toContain(
|
||||
"When a first-class tool exists for an action, use the tool instead of asking the user to run a command.",
|
||||
);
|
||||
expect(OPENAI_GPT5_TOOL_CALL_STYLE).not.toContain("/approve");
|
||||
expect(OPENAI_GPT5_OUTPUT_CONTRACT).toContain(
|
||||
"Return the requested sections only, in the requested order.",
|
||||
);
|
||||
expect(OPENAI_GPT5_OUTPUT_CONTRACT).toContain(
|
||||
"Prefer commas, periods, or parentheses over em dashes in normal prose.",
|
||||
);
|
||||
expect(OPENAI_GPT5_OUTPUT_CONTRACT).toContain(
|
||||
"Do not use em dashes unless the user explicitly asks for them or they are required in quoted text.",
|
||||
);
|
||||
});
|
||||
|
||||
it("defaults to the friendly OpenAI interaction-style overlay", async () => {
|
||||
const { on, providers } = await registerOpenAIPluginWithHook();
|
||||
|
||||
expect(on).not.toHaveBeenCalledWith("before_prompt_build", expect.any(Function));
|
||||
const openaiProvider = requireRegisteredProvider(providers, "openai");
|
||||
expect(
|
||||
openaiProvider.resolveSystemPromptContribution?.({
|
||||
config: undefined,
|
||||
agentDir: undefined,
|
||||
workspaceDir: undefined,
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.4",
|
||||
promptMode: "full",
|
||||
runtimeChannel: undefined,
|
||||
runtimeCapabilities: undefined,
|
||||
agentId: undefined,
|
||||
}),
|
||||
).toEqual({
|
||||
stablePrefix: [OPENAI_GPT5_OUTPUT_CONTRACT, OPENAI_GPT5_TOOL_CALL_STYLE].join("\n\n"),
|
||||
sectionOverrides: {
|
||||
interaction_style: OPENAI_FRIENDLY_PROMPT_OVERLAY,
|
||||
execution_bias: OPENAI_GPT5_EXECUTION_BIAS,
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it("supports opting out of the friendly prompt overlay via plugin config", async () => {
|
||||
const { on, providers } = await registerOpenAIPluginWithHook({
|
||||
pluginConfig: { personality: "off" },
|
||||
});
|
||||
|
||||
expect(on).not.toHaveBeenCalledWith("before_prompt_build", expect.any(Function));
|
||||
const openaiProvider = requireRegisteredProvider(providers, "openai");
|
||||
expect(
|
||||
openaiProvider.resolveSystemPromptContribution?.({
|
||||
config: undefined,
|
||||
agentDir: undefined,
|
||||
workspaceDir: undefined,
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.4",
|
||||
promptMode: "full",
|
||||
runtimeChannel: undefined,
|
||||
runtimeCapabilities: undefined,
|
||||
agentId: undefined,
|
||||
}),
|
||||
).toEqual({
|
||||
stablePrefix: [OPENAI_GPT5_OUTPUT_CONTRACT, OPENAI_GPT5_TOOL_CALL_STYLE].join("\n\n"),
|
||||
sectionOverrides: {
|
||||
execution_bias: OPENAI_GPT5_EXECUTION_BIAS,
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it("treats mixed-case off values as disabling the friendly prompt overlay", async () => {
|
||||
const { providers } = await registerOpenAIPluginWithHook({
|
||||
pluginConfig: { personality: "Off" },
|
||||
});
|
||||
|
||||
const openaiProvider = requireRegisteredProvider(providers, "openai");
|
||||
expect(
|
||||
openaiProvider.resolveSystemPromptContribution?.({
|
||||
config: undefined,
|
||||
agentDir: undefined,
|
||||
workspaceDir: undefined,
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.4",
|
||||
promptMode: "full",
|
||||
runtimeChannel: undefined,
|
||||
runtimeCapabilities: undefined,
|
||||
agentId: undefined,
|
||||
}),
|
||||
).toEqual({
|
||||
stablePrefix: [OPENAI_GPT5_OUTPUT_CONTRACT, OPENAI_GPT5_TOOL_CALL_STYLE].join("\n\n"),
|
||||
sectionOverrides: {
|
||||
execution_bias: OPENAI_GPT5_EXECUTION_BIAS,
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it("supports explicitly configuring the friendly prompt overlay", async () => {
|
||||
const { on, providers } = await registerOpenAIPluginWithHook({
|
||||
pluginConfig: { personality: "friendly" },
|
||||
});
|
||||
|
||||
expect(on).not.toHaveBeenCalledWith("before_prompt_build", expect.any(Function));
|
||||
const openaiProvider = requireRegisteredProvider(providers, "openai");
|
||||
expect(
|
||||
openaiProvider.resolveSystemPromptContribution?.({
|
||||
config: undefined,
|
||||
agentDir: undefined,
|
||||
workspaceDir: undefined,
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.4",
|
||||
promptMode: "full",
|
||||
runtimeChannel: undefined,
|
||||
runtimeCapabilities: undefined,
|
||||
agentId: undefined,
|
||||
}),
|
||||
).toEqual({
|
||||
stablePrefix: [OPENAI_GPT5_OUTPUT_CONTRACT, OPENAI_GPT5_TOOL_CALL_STYLE].join("\n\n"),
|
||||
sectionOverrides: {
|
||||
interaction_style: OPENAI_FRIENDLY_PROMPT_OVERLAY,
|
||||
execution_bias: OPENAI_GPT5_EXECUTION_BIAS,
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it("treats on as an alias for the friendly prompt overlay", async () => {
|
||||
const { providers } = await registerOpenAIPluginWithHook({
|
||||
pluginConfig: { personality: "on" },
|
||||
});
|
||||
|
||||
const openaiProvider = requireRegisteredProvider(providers, "openai");
|
||||
expect(
|
||||
openaiProvider.resolveSystemPromptContribution?.({
|
||||
config: undefined,
|
||||
agentDir: undefined,
|
||||
workspaceDir: undefined,
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.4",
|
||||
promptMode: "full",
|
||||
runtimeChannel: undefined,
|
||||
runtimeCapabilities: undefined,
|
||||
agentId: undefined,
|
||||
}),
|
||||
).toEqual({
|
||||
stablePrefix: [OPENAI_GPT5_OUTPUT_CONTRACT, OPENAI_GPT5_TOOL_CALL_STYLE].join("\n\n"),
|
||||
sectionOverrides: {
|
||||
interaction_style: OPENAI_FRIENDLY_PROMPT_OVERLAY,
|
||||
execution_bias: OPENAI_GPT5_EXECUTION_BIAS,
|
||||
},
|
||||
});
|
||||
});
|
||||
});
|
||||
52
openclaw/extensions/openai/index.ts
Normal file
52
openclaw/extensions/openai/index.ts
Normal file
|
|
@ -0,0 +1,52 @@
|
|||
import { definePluginEntry } from "openclaw/plugin-sdk/plugin-entry";
|
||||
import { buildProviderToolCompatFamilyHooks } from "openclaw/plugin-sdk/provider-tools";
|
||||
import { buildOpenAICodexCliBackend } from "./cli-backend.js";
|
||||
import { buildOpenAIImageGenerationProvider } from "./image-generation-provider.js";
|
||||
import {
|
||||
openaiCodexMediaUnderstandingProvider,
|
||||
openaiMediaUnderstandingProvider,
|
||||
} from "./media-understanding-provider.js";
|
||||
import { openAiMemoryEmbeddingProviderAdapter } from "./memory-embedding-adapter.js";
|
||||
import { buildOpenAICodexProviderPlugin } from "./openai-codex-provider.js";
|
||||
import { buildOpenAIProvider } from "./openai-provider.js";
|
||||
import {
|
||||
resolveOpenAIPromptOverlayMode,
|
||||
resolveOpenAISystemPromptContribution,
|
||||
} from "./prompt-overlay.js";
|
||||
import { buildOpenAIRealtimeTranscriptionProvider } from "./realtime-transcription-provider.js";
|
||||
import { buildOpenAIRealtimeVoiceProvider } from "./realtime-voice-provider.js";
|
||||
import { buildOpenAISpeechProvider } from "./speech-provider.js";
|
||||
import { buildOpenAIVideoGenerationProvider } from "./video-generation-provider.js";
|
||||
|
||||
export default definePluginEntry({
|
||||
id: "openai",
|
||||
name: "OpenAI Provider",
|
||||
description: "Bundled OpenAI provider plugins",
|
||||
register(api) {
|
||||
const promptOverlayMode = resolveOpenAIPromptOverlayMode(api.pluginConfig);
|
||||
const openAIToolCompatHooks = buildProviderToolCompatFamilyHooks("openai");
|
||||
const buildProviderWithPromptContribution = <T extends ReturnType<typeof buildOpenAIProvider>>(
|
||||
provider: T,
|
||||
): T => ({
|
||||
...provider,
|
||||
...openAIToolCompatHooks,
|
||||
resolveSystemPromptContribution: (ctx) =>
|
||||
resolveOpenAISystemPromptContribution({
|
||||
mode: promptOverlayMode,
|
||||
modelProviderId: provider.id,
|
||||
modelId: ctx.modelId,
|
||||
}),
|
||||
});
|
||||
api.registerCliBackend(buildOpenAICodexCliBackend());
|
||||
api.registerProvider(buildProviderWithPromptContribution(buildOpenAIProvider()));
|
||||
api.registerProvider(buildProviderWithPromptContribution(buildOpenAICodexProviderPlugin()));
|
||||
api.registerMemoryEmbeddingProvider(openAiMemoryEmbeddingProviderAdapter);
|
||||
api.registerImageGenerationProvider(buildOpenAIImageGenerationProvider());
|
||||
api.registerRealtimeTranscriptionProvider(buildOpenAIRealtimeTranscriptionProvider());
|
||||
api.registerRealtimeVoiceProvider(buildOpenAIRealtimeVoiceProvider());
|
||||
api.registerSpeechProvider(buildOpenAISpeechProvider());
|
||||
api.registerMediaUnderstandingProvider(openaiMediaUnderstandingProvider);
|
||||
api.registerMediaUnderstandingProvider(openaiCodexMediaUnderstandingProvider);
|
||||
api.registerVideoGenerationProvider(buildOpenAIVideoGenerationProvider());
|
||||
},
|
||||
});
|
||||
|
|
@ -0,0 +1,84 @@
|
|||
import { describe, expect, it } from "vitest";
|
||||
import {
|
||||
createAuthCaptureJsonFetch,
|
||||
createRequestCaptureJsonFetch,
|
||||
installPinnedHostnameTestHooks,
|
||||
} from "../../src/media-understanding/audio.test-helpers.ts";
|
||||
import { transcribeOpenAiAudio } from "./media-understanding-provider.js";
|
||||
|
||||
installPinnedHostnameTestHooks();
|
||||
|
||||
describe("transcribeOpenAiAudio", () => {
|
||||
it("respects lowercase authorization header overrides", async () => {
|
||||
const { fetchFn, getAuthHeader } = createAuthCaptureJsonFetch({ text: "ok" });
|
||||
|
||||
const result = await transcribeOpenAiAudio({
|
||||
buffer: Buffer.from("audio"),
|
||||
fileName: "note.mp3",
|
||||
apiKey: "test-key",
|
||||
timeoutMs: 1000,
|
||||
headers: { authorization: "Bearer override" },
|
||||
fetchFn,
|
||||
});
|
||||
|
||||
expect(getAuthHeader()).toBe("Bearer override");
|
||||
expect(result.text).toBe("ok");
|
||||
});
|
||||
|
||||
it("builds the expected request payload", async () => {
|
||||
const { fetchFn, getRequest } = createRequestCaptureJsonFetch({ text: "hello" });
|
||||
|
||||
const result = await transcribeOpenAiAudio({
|
||||
buffer: Buffer.from("audio-bytes"),
|
||||
fileName: "voice.wav",
|
||||
apiKey: "test-key",
|
||||
timeoutMs: 1234,
|
||||
baseUrl: "https://api.example.com/v1/",
|
||||
model: " ",
|
||||
language: " en ",
|
||||
prompt: " hello ",
|
||||
mime: "audio/wav",
|
||||
headers: { "X-Custom": "1" },
|
||||
fetchFn,
|
||||
});
|
||||
const { url: seenUrl, init: seenInit } = getRequest();
|
||||
|
||||
expect(result.model).toBe("gpt-4o-transcribe");
|
||||
expect(result.text).toBe("hello");
|
||||
expect(seenUrl).toBe("https://api.example.com/v1/audio/transcriptions");
|
||||
expect(seenInit?.method).toBe("POST");
|
||||
expect(seenInit?.signal).toBeInstanceOf(AbortSignal);
|
||||
|
||||
const headers = new Headers(seenInit?.headers);
|
||||
expect(headers.get("authorization")).toBe("Bearer test-key");
|
||||
expect(headers.get("x-custom")).toBe("1");
|
||||
|
||||
const form = seenInit?.body as FormData;
|
||||
expect(form).toBeInstanceOf(FormData);
|
||||
expect(form.get("model")).toBe("gpt-4o-transcribe");
|
||||
expect(form.get("language")).toBe("en");
|
||||
expect(form.get("prompt")).toBe("hello");
|
||||
const file = form.get("file") as Blob | { type?: string; name?: string } | null;
|
||||
expect(file).not.toBeNull();
|
||||
if (file) {
|
||||
expect(file.type).toBe("audio/wav");
|
||||
if ("name" in file && typeof file.name === "string") {
|
||||
expect(file.name).toBe("voice.wav");
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
it("throws when the provider response omits text", async () => {
|
||||
const { fetchFn } = createRequestCaptureJsonFetch({});
|
||||
|
||||
await expect(
|
||||
transcribeOpenAiAudio({
|
||||
buffer: Buffer.from("audio-bytes"),
|
||||
fileName: "voice.wav",
|
||||
apiKey: "test-key",
|
||||
timeoutMs: 1234,
|
||||
fetchFn,
|
||||
}),
|
||||
).rejects.toThrow("Audio transcription response missing text");
|
||||
});
|
||||
});
|
||||
40
openclaw/extensions/openai/media-understanding-provider.ts
Normal file
40
openclaw/extensions/openai/media-understanding-provider.ts
Normal file
|
|
@ -0,0 +1,40 @@
|
|||
import {
|
||||
describeImageWithModel,
|
||||
describeImagesWithModel,
|
||||
transcribeOpenAiCompatibleAudio,
|
||||
type AudioTranscriptionRequest,
|
||||
type MediaUnderstandingProvider,
|
||||
} from "openclaw/plugin-sdk/media-understanding";
|
||||
import { OPENAI_DEFAULT_AUDIO_TRANSCRIPTION_MODEL } from "./default-models.js";
|
||||
|
||||
export const DEFAULT_OPENAI_AUDIO_BASE_URL = "https://api.openai.com/v1";
|
||||
|
||||
export async function transcribeOpenAiAudio(params: AudioTranscriptionRequest) {
|
||||
return await transcribeOpenAiCompatibleAudio({
|
||||
...params,
|
||||
provider: "openai",
|
||||
defaultBaseUrl: DEFAULT_OPENAI_AUDIO_BASE_URL,
|
||||
defaultModel: OPENAI_DEFAULT_AUDIO_TRANSCRIPTION_MODEL,
|
||||
});
|
||||
}
|
||||
|
||||
export const openaiMediaUnderstandingProvider: MediaUnderstandingProvider = {
|
||||
id: "openai",
|
||||
capabilities: ["image", "audio"],
|
||||
defaultModels: {
|
||||
image: "gpt-5.4-mini",
|
||||
audio: OPENAI_DEFAULT_AUDIO_TRANSCRIPTION_MODEL,
|
||||
},
|
||||
autoPriority: { image: 10, audio: 10 },
|
||||
describeImage: describeImageWithModel,
|
||||
describeImages: describeImagesWithModel,
|
||||
transcribeAudio: transcribeOpenAiAudio,
|
||||
};
|
||||
|
||||
export const openaiCodexMediaUnderstandingProvider: MediaUnderstandingProvider = {
|
||||
id: "openai-codex",
|
||||
capabilities: ["image"],
|
||||
defaultModels: { image: "gpt-5.4" },
|
||||
describeImage: describeImageWithModel,
|
||||
describeImages: describeImagesWithModel,
|
||||
};
|
||||
61
openclaw/extensions/openai/memory-embedding-adapter.ts
Normal file
61
openclaw/extensions/openai/memory-embedding-adapter.ts
Normal file
|
|
@ -0,0 +1,61 @@
|
|||
import {
|
||||
isMissingEmbeddingApiKeyError,
|
||||
mapBatchEmbeddingsByIndex,
|
||||
sanitizeEmbeddingCacheHeaders,
|
||||
type MemoryEmbeddingProviderAdapter,
|
||||
} from "openclaw/plugin-sdk/memory-core-host-engine-embeddings";
|
||||
import { OPENAI_BATCH_ENDPOINT, runOpenAiEmbeddingBatches } from "./embedding-batch.js";
|
||||
import {
|
||||
createOpenAiEmbeddingProvider,
|
||||
DEFAULT_OPENAI_EMBEDDING_MODEL,
|
||||
} from "./embedding-provider.js";
|
||||
|
||||
export const openAiMemoryEmbeddingProviderAdapter: MemoryEmbeddingProviderAdapter = {
|
||||
id: "openai",
|
||||
defaultModel: DEFAULT_OPENAI_EMBEDDING_MODEL,
|
||||
transport: "remote",
|
||||
authProviderId: "openai",
|
||||
autoSelectPriority: 20,
|
||||
allowExplicitWhenConfiguredAuto: true,
|
||||
shouldContinueAutoSelection: isMissingEmbeddingApiKeyError,
|
||||
create: async (options) => {
|
||||
const { provider, client } = await createOpenAiEmbeddingProvider({
|
||||
...options,
|
||||
provider: "openai",
|
||||
fallback: "none",
|
||||
});
|
||||
return {
|
||||
provider,
|
||||
runtime: {
|
||||
id: "openai",
|
||||
cacheKeyData: {
|
||||
provider: "openai",
|
||||
baseUrl: client.baseUrl,
|
||||
model: client.model,
|
||||
headers: sanitizeEmbeddingCacheHeaders(client.headers, ["authorization"]),
|
||||
},
|
||||
batchEmbed: async (batch) => {
|
||||
const byCustomId = await runOpenAiEmbeddingBatches({
|
||||
openAi: client,
|
||||
agentId: batch.agentId,
|
||||
requests: batch.chunks.map((chunk, index) => ({
|
||||
custom_id: String(index),
|
||||
method: "POST",
|
||||
url: OPENAI_BATCH_ENDPOINT,
|
||||
body: {
|
||||
model: client.model,
|
||||
input: chunk.text,
|
||||
},
|
||||
})),
|
||||
wait: batch.wait,
|
||||
concurrency: batch.concurrency,
|
||||
pollIntervalMs: batch.pollIntervalMs,
|
||||
timeoutMs: batch.timeoutMs,
|
||||
debug: batch.debug,
|
||||
});
|
||||
return mapBatchEmbeddingsByIndex(byCustomId, batch.chunks.length);
|
||||
},
|
||||
},
|
||||
};
|
||||
},
|
||||
};
|
||||
|
|
@ -0,0 +1,56 @@
|
|||
import { describe, expect, it } from "vitest";
|
||||
import { resolveCodexAuthIdentity } from "./openai-codex-auth-identity.js";
|
||||
|
||||
function createJwt(payload: Record<string, unknown>): string {
|
||||
const header = Buffer.from(JSON.stringify({ alg: "none", typ: "JWT" })).toString("base64url");
|
||||
const body = Buffer.from(JSON.stringify(payload)).toString("base64url");
|
||||
return `${header}.${body}.signature`;
|
||||
}
|
||||
|
||||
describe("resolveCodexAuthIdentity", () => {
|
||||
it("prefers JWT profile email when present", () => {
|
||||
const identity = resolveCodexAuthIdentity({
|
||||
accessToken: createJwt({
|
||||
"https://api.openai.com/profile": {
|
||||
email: "jwt-user@example.com",
|
||||
},
|
||||
}),
|
||||
email: "credential@example.com",
|
||||
});
|
||||
|
||||
expect(identity).toEqual({
|
||||
email: "jwt-user@example.com",
|
||||
profileName: "jwt-user@example.com",
|
||||
});
|
||||
});
|
||||
|
||||
it("falls back to credential email before synthetic ids", () => {
|
||||
const identity = resolveCodexAuthIdentity({
|
||||
accessToken: createJwt({}),
|
||||
email: "credential@example.com",
|
||||
});
|
||||
|
||||
expect(identity).toEqual({
|
||||
email: "credential@example.com",
|
||||
profileName: "credential@example.com",
|
||||
});
|
||||
});
|
||||
|
||||
it("derives a stable profile id when email is missing", () => {
|
||||
const identity = resolveCodexAuthIdentity({
|
||||
accessToken: createJwt({
|
||||
"https://api.openai.com/auth": {
|
||||
chatgpt_account_user_id: "user-123__acct-456",
|
||||
},
|
||||
}),
|
||||
});
|
||||
|
||||
expect(identity).toEqual({
|
||||
profileName: `id-${Buffer.from("user-123__acct-456").toString("base64url")}`,
|
||||
});
|
||||
});
|
||||
|
||||
it("returns no metadata when token parsing yields no identity", () => {
|
||||
expect(resolveCodexAuthIdentity({ accessToken: "not-a-jwt-token" })).toEqual({});
|
||||
});
|
||||
});
|
||||
88
openclaw/extensions/openai/openai-codex-auth-identity.ts
Normal file
88
openclaw/extensions/openai/openai-codex-auth-identity.ts
Normal file
|
|
@ -0,0 +1,88 @@
|
|||
import { trimNonEmptyString } from "./openai-codex-shared.js";
|
||||
|
||||
type CodexJwtPayload = {
|
||||
exp?: unknown;
|
||||
iss?: unknown;
|
||||
sub?: unknown;
|
||||
"https://api.openai.com/profile"?: {
|
||||
email?: unknown;
|
||||
};
|
||||
"https://api.openai.com/auth"?: {
|
||||
chatgpt_account_user_id?: unknown;
|
||||
chatgpt_user_id?: unknown;
|
||||
user_id?: unknown;
|
||||
};
|
||||
};
|
||||
|
||||
function normalizeFutureEpochSeconds(value: unknown): number | undefined {
|
||||
if (typeof value === "number" && Number.isFinite(value) && value > 0) {
|
||||
return Math.trunc(value);
|
||||
}
|
||||
if (typeof value === "string" && /^\d+$/.test(value.trim())) {
|
||||
return Number.parseInt(value.trim(), 10);
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
export function decodeCodexJwtPayload(accessToken: string): CodexJwtPayload | null {
|
||||
const parts = accessToken.split(".");
|
||||
if (parts.length !== 3) {
|
||||
return null;
|
||||
}
|
||||
|
||||
try {
|
||||
const decoded = Buffer.from(parts[1], "base64url").toString("utf8");
|
||||
const parsed = JSON.parse(decoded);
|
||||
return parsed && typeof parsed === "object" ? (parsed as CodexJwtPayload) : null;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
export function resolveCodexStableSubject(payload: CodexJwtPayload | null): string | undefined {
|
||||
const auth = payload?.["https://api.openai.com/auth"];
|
||||
const accountUserId = trimNonEmptyString(auth?.chatgpt_account_user_id);
|
||||
if (accountUserId) {
|
||||
return accountUserId;
|
||||
}
|
||||
|
||||
const userId = trimNonEmptyString(auth?.chatgpt_user_id) ?? trimNonEmptyString(auth?.user_id);
|
||||
if (userId) {
|
||||
return userId;
|
||||
}
|
||||
|
||||
const iss = trimNonEmptyString(payload?.iss);
|
||||
const sub = trimNonEmptyString(payload?.sub);
|
||||
if (iss && sub) {
|
||||
return `${iss}|${sub}`;
|
||||
}
|
||||
return sub;
|
||||
}
|
||||
|
||||
export function resolveCodexAccessTokenExpiry(accessToken: string): number | undefined {
|
||||
const payload = decodeCodexJwtPayload(accessToken);
|
||||
const exp = normalizeFutureEpochSeconds(payload?.exp);
|
||||
return exp ? exp * 1000 : undefined;
|
||||
}
|
||||
|
||||
export function resolveCodexAuthIdentity(params: { accessToken: string; email?: string | null }): {
|
||||
email?: string;
|
||||
profileName?: string;
|
||||
} {
|
||||
const payload = decodeCodexJwtPayload(params.accessToken);
|
||||
const email =
|
||||
trimNonEmptyString(payload?.["https://api.openai.com/profile"]?.email) ??
|
||||
trimNonEmptyString(params.email);
|
||||
if (email) {
|
||||
return { email, profileName: email };
|
||||
}
|
||||
|
||||
const stableSubject = resolveCodexStableSubject(payload);
|
||||
if (!stableSubject) {
|
||||
return {};
|
||||
}
|
||||
|
||||
return {
|
||||
profileName: `id-${Buffer.from(stableSubject).toString("base64url")}`,
|
||||
};
|
||||
}
|
||||
11
openclaw/extensions/openai/openai-codex-catalog.ts
Normal file
11
openclaw/extensions/openai/openai-codex-catalog.ts
Normal file
|
|
@ -0,0 +1,11 @@
|
|||
import type { ModelProviderConfig } from "openclaw/plugin-sdk/provider-model-shared";
|
||||
|
||||
export const OPENAI_CODEX_BASE_URL = "https://chatgpt.com/backend-api";
|
||||
|
||||
export function buildOpenAICodexProvider(): ModelProviderConfig {
|
||||
return {
|
||||
baseUrl: OPENAI_CODEX_BASE_URL,
|
||||
api: "openai-codex-responses",
|
||||
models: [],
|
||||
};
|
||||
}
|
||||
345
openclaw/extensions/openai/openai-codex-cli-auth.test.ts
Normal file
345
openclaw/extensions/openai/openai-codex-cli-auth.test.ts
Normal file
|
|
@ -0,0 +1,345 @@
|
|||
import fs from "node:fs";
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
|
||||
|
||||
const runtimeMocks = vi.hoisted(() => ({
|
||||
debug: vi.fn(),
|
||||
}));
|
||||
|
||||
vi.mock("openclaw/plugin-sdk/runtime-env", () => ({
|
||||
createSubsystemLogger: () => ({
|
||||
debug: runtimeMocks.debug,
|
||||
}),
|
||||
}));
|
||||
|
||||
import {
|
||||
OPENAI_CODEX_DEFAULT_PROFILE_ID,
|
||||
readOpenAICodexCliOAuthProfile,
|
||||
} from "./openai-codex-cli-auth.js";
|
||||
|
||||
function buildJwt(payload: Record<string, unknown>) {
|
||||
const encode = (value: Record<string, unknown>) =>
|
||||
Buffer.from(JSON.stringify(value)).toString("base64url");
|
||||
return `${encode({ alg: "none", typ: "JWT" })}.${encode(payload)}.sig`;
|
||||
}
|
||||
|
||||
describe("readOpenAICodexCliOAuthProfile", () => {
|
||||
beforeEach(() => {
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
it("reads Codex CLI chatgpt auth into the default OpenAI Codex profile", () => {
|
||||
const accessToken = buildJwt({
|
||||
exp: Math.floor(Date.now() / 1000) + 600,
|
||||
"https://api.openai.com/profile": {
|
||||
email: "codex@example.com",
|
||||
},
|
||||
});
|
||||
vi.spyOn(fs, "readFileSync").mockReturnValue(
|
||||
JSON.stringify({
|
||||
auth_mode: "chatgpt",
|
||||
tokens: {
|
||||
id_token: "id-token",
|
||||
access_token: accessToken,
|
||||
refresh_token: "refresh-token",
|
||||
account_id: "acct_123",
|
||||
},
|
||||
}),
|
||||
);
|
||||
|
||||
const parsed = readOpenAICodexCliOAuthProfile({
|
||||
store: { version: 1, profiles: {} },
|
||||
});
|
||||
|
||||
expect(parsed).toMatchObject({
|
||||
profileId: OPENAI_CODEX_DEFAULT_PROFILE_ID,
|
||||
credential: {
|
||||
type: "oauth",
|
||||
provider: "openai-codex",
|
||||
access: accessToken,
|
||||
refresh: "refresh-token",
|
||||
accountId: "acct_123",
|
||||
idToken: "id-token",
|
||||
email: "codex@example.com",
|
||||
},
|
||||
});
|
||||
expect(parsed?.credential.expires).toBeGreaterThan(Date.now());
|
||||
});
|
||||
|
||||
it("does not override a locally managed OpenAI Codex profile", () => {
|
||||
vi.spyOn(fs, "readFileSync").mockReturnValue(
|
||||
JSON.stringify({
|
||||
auth_mode: "chatgpt",
|
||||
tokens: {
|
||||
access_token: "access-token",
|
||||
refresh_token: "refresh-token",
|
||||
},
|
||||
}),
|
||||
);
|
||||
|
||||
const parsed = readOpenAICodexCliOAuthProfile({
|
||||
store: {
|
||||
version: 1,
|
||||
profiles: {
|
||||
[OPENAI_CODEX_DEFAULT_PROFILE_ID]: {
|
||||
type: "oauth",
|
||||
provider: "openai-codex",
|
||||
access: "local-access",
|
||||
refresh: "local-refresh",
|
||||
expires: Date.now() + 10 * 60_000,
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(parsed).toBeNull();
|
||||
});
|
||||
|
||||
it("does not override explicit local non-oauth auth with Codex CLI bootstrap", () => {
|
||||
vi.spyOn(fs, "readFileSync").mockReturnValue(
|
||||
JSON.stringify({
|
||||
auth_mode: "chatgpt",
|
||||
tokens: {
|
||||
access_token: "access-token",
|
||||
refresh_token: "refresh-token",
|
||||
},
|
||||
}),
|
||||
);
|
||||
|
||||
const parsed = readOpenAICodexCliOAuthProfile({
|
||||
store: {
|
||||
version: 1,
|
||||
profiles: {
|
||||
[OPENAI_CODEX_DEFAULT_PROFILE_ID]: {
|
||||
type: "api_key",
|
||||
provider: "openai-codex",
|
||||
key: "sk-local",
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(parsed).toBeNull();
|
||||
});
|
||||
|
||||
it("refuses Codex CLI bootstrap when an expired local default belongs to a different account", () => {
|
||||
const accessToken = buildJwt({
|
||||
exp: Math.floor(Date.now() / 1000) + 600,
|
||||
"https://api.openai.com/profile": {
|
||||
email: "codex-b@example.com",
|
||||
},
|
||||
});
|
||||
vi.spyOn(fs, "readFileSync").mockReturnValue(
|
||||
JSON.stringify({
|
||||
auth_mode: "chatgpt",
|
||||
tokens: {
|
||||
access_token: accessToken,
|
||||
refresh_token: "refresh-token",
|
||||
account_id: "acct_b",
|
||||
},
|
||||
}),
|
||||
);
|
||||
|
||||
const parsed = readOpenAICodexCliOAuthProfile({
|
||||
store: {
|
||||
version: 1,
|
||||
profiles: {
|
||||
[OPENAI_CODEX_DEFAULT_PROFILE_ID]: {
|
||||
type: "oauth",
|
||||
provider: "openai-codex",
|
||||
access: "near-expiry-local-access",
|
||||
refresh: "near-expiry-local-refresh",
|
||||
expires: Date.now() + 60_000,
|
||||
accountId: "acct_a",
|
||||
email: "codex-a@example.com",
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(parsed).toBeNull();
|
||||
});
|
||||
|
||||
it("allows cli bootstrap when the stored default profile is expired", () => {
|
||||
const accessToken = buildJwt({
|
||||
exp: Math.floor(Date.now() / 1000) + 600,
|
||||
"https://api.openai.com/profile": {
|
||||
email: "codex@example.com",
|
||||
},
|
||||
});
|
||||
vi.spyOn(fs, "readFileSync").mockReturnValue(
|
||||
JSON.stringify({
|
||||
auth_mode: "chatgpt",
|
||||
tokens: {
|
||||
id_token: "id-token",
|
||||
access_token: accessToken,
|
||||
refresh_token: "refresh-token",
|
||||
account_id: "acct_123",
|
||||
},
|
||||
}),
|
||||
);
|
||||
|
||||
const parsed = readOpenAICodexCliOAuthProfile({
|
||||
store: {
|
||||
version: 1,
|
||||
profiles: {
|
||||
[OPENAI_CODEX_DEFAULT_PROFILE_ID]: {
|
||||
type: "oauth",
|
||||
provider: "openai-codex",
|
||||
access: "expired-local-access",
|
||||
refresh: "expired-local-refresh",
|
||||
expires: Date.now() - 60_000,
|
||||
accountId: "acct_123",
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(parsed).toMatchObject({
|
||||
profileId: OPENAI_CODEX_DEFAULT_PROFILE_ID,
|
||||
credential: {
|
||||
access: accessToken,
|
||||
refresh: "refresh-token",
|
||||
accountId: "acct_123",
|
||||
idToken: "id-token",
|
||||
email: "codex@example.com",
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it("refuses cli bootstrap when the stored default profile is expired but identity mismatches", () => {
|
||||
const accessToken = buildJwt({
|
||||
exp: Math.floor(Date.now() / 1000) + 600,
|
||||
"https://api.openai.com/profile": {
|
||||
email: "codex@example.com",
|
||||
},
|
||||
});
|
||||
vi.spyOn(fs, "readFileSync").mockReturnValue(
|
||||
JSON.stringify({
|
||||
auth_mode: "chatgpt",
|
||||
tokens: {
|
||||
id_token: "id-token",
|
||||
access_token: accessToken,
|
||||
refresh_token: "refresh-token",
|
||||
account_id: "acct_123",
|
||||
},
|
||||
}),
|
||||
);
|
||||
|
||||
const parsed = readOpenAICodexCliOAuthProfile({
|
||||
store: {
|
||||
version: 1,
|
||||
profiles: {
|
||||
[OPENAI_CODEX_DEFAULT_PROFILE_ID]: {
|
||||
type: "oauth",
|
||||
provider: "openai-codex",
|
||||
access: "expired-local-access",
|
||||
refresh: "expired-local-refresh",
|
||||
expires: Date.now() - 60_000,
|
||||
accountId: "acct_local",
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(parsed).toBeNull();
|
||||
});
|
||||
|
||||
it("allows the runtime-only Codex CLI profile when the stored default already matches", () => {
|
||||
const accessToken = buildJwt({
|
||||
exp: Math.floor(Date.now() / 1000) + 600,
|
||||
"https://api.openai.com/profile": {
|
||||
email: "codex@example.com",
|
||||
},
|
||||
});
|
||||
vi.spyOn(fs, "readFileSync").mockReturnValue(
|
||||
JSON.stringify({
|
||||
auth_mode: "chatgpt",
|
||||
tokens: {
|
||||
id_token: "id-token",
|
||||
access_token: accessToken,
|
||||
refresh_token: "refresh-token",
|
||||
account_id: "acct_123",
|
||||
},
|
||||
}),
|
||||
);
|
||||
|
||||
const firstParse = readOpenAICodexCliOAuthProfile({
|
||||
store: { version: 1, profiles: {} },
|
||||
});
|
||||
expect(firstParse).not.toBeNull();
|
||||
|
||||
const parsed = readOpenAICodexCliOAuthProfile({
|
||||
store: {
|
||||
version: 1,
|
||||
profiles: {
|
||||
[OPENAI_CODEX_DEFAULT_PROFILE_ID]: firstParse!.credential,
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(parsed).toMatchObject({
|
||||
profileId: OPENAI_CODEX_DEFAULT_PROFILE_ID,
|
||||
credential: {
|
||||
access: accessToken,
|
||||
refresh: "refresh-token",
|
||||
accountId: "acct_123",
|
||||
idToken: "id-token",
|
||||
email: "codex@example.com",
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it("returns null without logging when the Codex CLI auth file is missing", () => {
|
||||
const error = Object.assign(new Error("missing"), {
|
||||
code: "ENOENT",
|
||||
});
|
||||
vi.spyOn(fs, "readFileSync").mockImplementation(() => {
|
||||
throw error;
|
||||
});
|
||||
|
||||
const parsed = readOpenAICodexCliOAuthProfile({
|
||||
store: { version: 1, profiles: {} },
|
||||
});
|
||||
|
||||
expect(parsed).toBeNull();
|
||||
expect(runtimeMocks.debug).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("logs a sanitized code for invalid auth JSON", () => {
|
||||
vi.spyOn(fs, "readFileSync").mockReturnValue("{");
|
||||
|
||||
const parsed = readOpenAICodexCliOAuthProfile({
|
||||
store: { version: 1, profiles: {} },
|
||||
});
|
||||
|
||||
expect(parsed).toBeNull();
|
||||
expect(runtimeMocks.debug).toHaveBeenCalledWith(
|
||||
"Failed to read Codex CLI auth file (code=INVALID_JSON)",
|
||||
);
|
||||
});
|
||||
|
||||
it("does not leak auth file paths in debug logs for filesystem failures", () => {
|
||||
const error = Object.assign(
|
||||
new Error("EACCES: permission denied, open '/Users/alice/.codex/auth.json'"),
|
||||
{
|
||||
code: "EACCES",
|
||||
},
|
||||
);
|
||||
vi.spyOn(fs, "readFileSync").mockImplementation(() => {
|
||||
throw error;
|
||||
});
|
||||
|
||||
const parsed = readOpenAICodexCliOAuthProfile({
|
||||
store: { version: 1, profiles: {} },
|
||||
});
|
||||
|
||||
expect(parsed).toBeNull();
|
||||
expect(runtimeMocks.debug).toHaveBeenCalledWith(
|
||||
"Failed to read Codex CLI auth file (code=EACCES)",
|
||||
);
|
||||
});
|
||||
});
|
||||
175
openclaw/extensions/openai/openai-codex-cli-auth.ts
Normal file
175
openclaw/extensions/openai/openai-codex-cli-auth.ts
Normal file
|
|
@ -0,0 +1,175 @@
|
|||
import fs from "node:fs";
|
||||
import path from "node:path";
|
||||
import {
|
||||
hasUsableOAuthCredential,
|
||||
resolveRequiredHomeDir,
|
||||
type AuthProfileStore,
|
||||
type OAuthCredential,
|
||||
} from "openclaw/plugin-sdk/provider-auth";
|
||||
import { createSubsystemLogger } from "openclaw/plugin-sdk/runtime-env";
|
||||
import {
|
||||
resolveCodexAccessTokenExpiry,
|
||||
resolveCodexAuthIdentity,
|
||||
} from "./openai-codex-auth-identity.js";
|
||||
import { trimNonEmptyString } from "./openai-codex-shared.js";
|
||||
|
||||
const PROVIDER_ID = "openai-codex";
|
||||
const log = createSubsystemLogger("openai/codex-cli-auth");
|
||||
|
||||
export const CODEX_CLI_PROFILE_ID = `${PROVIDER_ID}:codex-cli`;
|
||||
export const OPENAI_CODEX_DEFAULT_PROFILE_ID = `${PROVIDER_ID}:default`;
|
||||
|
||||
type CodexCliAuthFile = {
|
||||
auth_mode?: unknown;
|
||||
tokens?: {
|
||||
id_token?: unknown;
|
||||
access_token?: unknown;
|
||||
refresh_token?: unknown;
|
||||
account_id?: unknown;
|
||||
};
|
||||
};
|
||||
|
||||
function resolveCodexCliHome(env: NodeJS.ProcessEnv): string {
|
||||
const configured = trimNonEmptyString(env.CODEX_HOME);
|
||||
if (!configured) {
|
||||
return path.join(resolveRequiredHomeDir(), ".codex");
|
||||
}
|
||||
if (configured === "~") {
|
||||
return resolveRequiredHomeDir();
|
||||
}
|
||||
if (configured.startsWith("~/")) {
|
||||
return path.join(resolveRequiredHomeDir(), configured.slice(2));
|
||||
}
|
||||
return path.resolve(configured);
|
||||
}
|
||||
|
||||
function readCodexCliAuthFile(env: NodeJS.ProcessEnv): CodexCliAuthFile | null {
|
||||
try {
|
||||
const authPath = path.join(resolveCodexCliHome(env), "auth.json");
|
||||
const raw = fs.readFileSync(authPath, "utf8");
|
||||
const parsed = JSON.parse(raw);
|
||||
return parsed && typeof parsed === "object" ? (parsed as CodexCliAuthFile) : null;
|
||||
} catch (error) {
|
||||
const code =
|
||||
error instanceof SyntaxError
|
||||
? "INVALID_JSON"
|
||||
: error instanceof Error && "code" in error
|
||||
? (error as NodeJS.ErrnoException).code
|
||||
: undefined;
|
||||
if (code === "ENOENT") {
|
||||
return null;
|
||||
}
|
||||
log.debug(
|
||||
`Failed to read Codex CLI auth file (code=${typeof code === "string" ? code : "UNKNOWN"})`,
|
||||
);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
function oauthCredentialMatches(a: OAuthCredential, b: OAuthCredential): boolean {
|
||||
return (
|
||||
a.type === b.type &&
|
||||
a.provider === b.provider &&
|
||||
a.access === b.access &&
|
||||
a.refresh === b.refresh &&
|
||||
a.clientId === b.clientId &&
|
||||
a.email === b.email &&
|
||||
a.displayName === b.displayName &&
|
||||
a.enterpriseUrl === b.enterpriseUrl &&
|
||||
a.projectId === b.projectId &&
|
||||
a.accountId === b.accountId &&
|
||||
a.idToken === b.idToken
|
||||
);
|
||||
}
|
||||
|
||||
function normalizeAuthIdentityToken(value: string | undefined): string | undefined {
|
||||
const trimmed = value?.trim();
|
||||
return trimmed ? trimmed : undefined;
|
||||
}
|
||||
|
||||
function normalizeAuthEmailToken(value: string | undefined): string | undefined {
|
||||
return normalizeAuthIdentityToken(value)?.toLowerCase();
|
||||
}
|
||||
|
||||
function hasIdentityContinuity(
|
||||
existing: Pick<OAuthCredential, "accountId" | "email"> | undefined,
|
||||
incoming: OAuthCredential,
|
||||
): boolean {
|
||||
if (!existing) {
|
||||
return true;
|
||||
}
|
||||
if (oauthCredentialMatches(existing as OAuthCredential, incoming)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
const existingAccountId = normalizeAuthIdentityToken(existing.accountId);
|
||||
const incomingAccountId = normalizeAuthIdentityToken(incoming.accountId);
|
||||
if (existingAccountId !== undefined && incomingAccountId !== undefined) {
|
||||
return existingAccountId === incomingAccountId;
|
||||
}
|
||||
|
||||
const existingEmail = normalizeAuthEmailToken(existing.email);
|
||||
const incomingEmail = normalizeAuthEmailToken(incoming.email);
|
||||
if (existingEmail !== undefined && incomingEmail !== undefined) {
|
||||
return existingEmail === incomingEmail;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
export function readOpenAICodexCliOAuthProfile(params: {
|
||||
env?: NodeJS.ProcessEnv;
|
||||
store: AuthProfileStore;
|
||||
}): { profileId: string; credential: OAuthCredential } | null {
|
||||
const authFile = readCodexCliAuthFile(params.env ?? process.env);
|
||||
if (!authFile || authFile.auth_mode !== "chatgpt") {
|
||||
return null;
|
||||
}
|
||||
|
||||
const access = trimNonEmptyString(authFile.tokens?.access_token);
|
||||
const refresh = trimNonEmptyString(authFile.tokens?.refresh_token);
|
||||
if (!access || !refresh) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const accountId = trimNonEmptyString(authFile.tokens?.account_id);
|
||||
const idToken = trimNonEmptyString(authFile.tokens?.id_token);
|
||||
const identity = resolveCodexAuthIdentity({ accessToken: access });
|
||||
const credential: OAuthCredential = {
|
||||
type: "oauth",
|
||||
provider: PROVIDER_ID,
|
||||
access,
|
||||
refresh,
|
||||
expires: resolveCodexAccessTokenExpiry(access) ?? 0,
|
||||
...(accountId ? { accountId } : {}),
|
||||
...(idToken ? { idToken } : {}),
|
||||
...(identity.email ? { email: identity.email } : {}),
|
||||
...(identity.profileName ? { displayName: identity.profileName } : {}),
|
||||
};
|
||||
const existing = params.store.profiles[OPENAI_CODEX_DEFAULT_PROFILE_ID];
|
||||
const existingOAuth =
|
||||
existing?.type === "oauth" && existing.provider === PROVIDER_ID ? existing : undefined;
|
||||
if (existing && !existingOAuth) {
|
||||
log.debug("kept explicit local auth over Codex CLI bootstrap", {
|
||||
profileId: OPENAI_CODEX_DEFAULT_PROFILE_ID,
|
||||
localType: existing.type,
|
||||
localProvider: existing.provider,
|
||||
});
|
||||
return null;
|
||||
}
|
||||
if (!hasIdentityContinuity(existingOAuth, credential)) {
|
||||
return null;
|
||||
}
|
||||
if (
|
||||
existingOAuth &&
|
||||
hasUsableOAuthCredential(existingOAuth) &&
|
||||
!oauthCredentialMatches(existingOAuth, credential)
|
||||
) {
|
||||
return null;
|
||||
}
|
||||
|
||||
return {
|
||||
profileId: OPENAI_CODEX_DEFAULT_PROFILE_ID,
|
||||
credential,
|
||||
};
|
||||
}
|
||||
145
openclaw/extensions/openai/openai-codex-cli-bridge.test.ts
Normal file
145
openclaw/extensions/openai/openai-codex-cli-bridge.test.ts
Normal file
|
|
@ -0,0 +1,145 @@
|
|||
import crypto from "node:crypto";
|
||||
import fs from "node:fs/promises";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
import { saveAuthProfileStore } from "openclaw/plugin-sdk/agent-runtime";
|
||||
import { afterEach, describe, expect, it } from "vitest";
|
||||
import { prepareOpenAICodexCliExecution } from "./openai-codex-cli-bridge.js";
|
||||
|
||||
describe("prepareOpenAICodexCliExecution", () => {
|
||||
const tempDirs: string[] = [];
|
||||
const resolveHashedCodexHome = (agentDir: string, profileId: string) =>
|
||||
path.join(
|
||||
agentDir,
|
||||
"cli-auth",
|
||||
"codex",
|
||||
crypto.createHash("sha256").update(profileId).digest("hex").slice(0, 16),
|
||||
);
|
||||
|
||||
afterEach(async () => {
|
||||
await Promise.all(
|
||||
tempDirs.splice(0).map((dir) => fs.rm(dir, { recursive: true, force: true })),
|
||||
);
|
||||
});
|
||||
|
||||
it("writes a private CODEX_HOME bridge from canonical OpenClaw oauth", async () => {
|
||||
const agentDir = await fs.mkdtemp(path.join(os.tmpdir(), "openclaw-codex-cli-bridge-"));
|
||||
tempDirs.push(agentDir);
|
||||
saveAuthProfileStore(
|
||||
{
|
||||
version: 1,
|
||||
profiles: {
|
||||
"openai-codex:default": {
|
||||
type: "oauth",
|
||||
provider: "openai-codex",
|
||||
access: "access-token",
|
||||
refresh: "refresh-token",
|
||||
expires: Date.now() + 60_000,
|
||||
accountId: "acct-123",
|
||||
idToken: "id-token",
|
||||
},
|
||||
},
|
||||
},
|
||||
agentDir,
|
||||
{ filterExternalAuthProfiles: false },
|
||||
);
|
||||
|
||||
const result = await prepareOpenAICodexCliExecution({
|
||||
config: undefined,
|
||||
workspaceDir: agentDir,
|
||||
agentDir,
|
||||
provider: "codex-cli",
|
||||
modelId: "gpt-5.4",
|
||||
authProfileId: "openai-codex:default",
|
||||
});
|
||||
|
||||
expect(result).toMatchObject({
|
||||
env: {
|
||||
CODEX_HOME: expect.stringContaining(path.join(agentDir, "cli-auth", "codex")),
|
||||
},
|
||||
clearEnv: ["OPENAI_API_KEY"],
|
||||
});
|
||||
|
||||
const authFile = JSON.parse(
|
||||
await fs.readFile(path.join(result?.env?.CODEX_HOME ?? "", "auth.json"), "utf8"),
|
||||
);
|
||||
expect(authFile).toEqual({
|
||||
auth_mode: "chatgpt",
|
||||
tokens: {
|
||||
id_token: "id-token",
|
||||
access_token: "access-token",
|
||||
refresh_token: "refresh-token",
|
||||
account_id: "acct-123",
|
||||
},
|
||||
});
|
||||
if (process.platform !== "win32") {
|
||||
const authStat = await fs.stat(path.join(result?.env?.CODEX_HOME ?? "", "auth.json"));
|
||||
expect(authStat.mode & 0o777).toBe(0o600);
|
||||
}
|
||||
});
|
||||
|
||||
it("returns null when there is no bridgeable canonical oauth credential", async () => {
|
||||
const agentDir = await fs.mkdtemp(path.join(os.tmpdir(), "openclaw-codex-cli-bridge-"));
|
||||
tempDirs.push(agentDir);
|
||||
saveAuthProfileStore(
|
||||
{
|
||||
version: 1,
|
||||
profiles: {
|
||||
"openai-codex:default": {
|
||||
type: "api_key",
|
||||
provider: "openai-codex",
|
||||
key: "sk-test",
|
||||
},
|
||||
},
|
||||
},
|
||||
agentDir,
|
||||
{ filterExternalAuthProfiles: false },
|
||||
);
|
||||
|
||||
await expect(
|
||||
prepareOpenAICodexCliExecution({
|
||||
config: undefined,
|
||||
workspaceDir: agentDir,
|
||||
agentDir,
|
||||
provider: "codex-cli",
|
||||
modelId: "gpt-5.4",
|
||||
authProfileId: "openai-codex:default",
|
||||
}),
|
||||
).resolves.toBeNull();
|
||||
});
|
||||
|
||||
it("refuses to overwrite a symlinked codex cli auth bridge file", async () => {
|
||||
const agentDir = await fs.mkdtemp(path.join(os.tmpdir(), "openclaw-codex-cli-bridge-"));
|
||||
tempDirs.push(agentDir);
|
||||
const codexHome = resolveHashedCodexHome(agentDir, "openai-codex:default");
|
||||
await fs.mkdir(codexHome, { recursive: true });
|
||||
await fs.symlink(path.join(agentDir, "outside.txt"), path.join(codexHome, "auth.json"));
|
||||
saveAuthProfileStore(
|
||||
{
|
||||
version: 1,
|
||||
profiles: {
|
||||
"openai-codex:default": {
|
||||
type: "oauth",
|
||||
provider: "openai-codex",
|
||||
access: "access-token",
|
||||
refresh: "refresh-token",
|
||||
expires: Date.now() + 60_000,
|
||||
},
|
||||
},
|
||||
},
|
||||
agentDir,
|
||||
{ filterExternalAuthProfiles: false },
|
||||
);
|
||||
|
||||
await expect(
|
||||
prepareOpenAICodexCliExecution({
|
||||
config: undefined,
|
||||
workspaceDir: agentDir,
|
||||
agentDir,
|
||||
provider: "codex-cli",
|
||||
modelId: "gpt-5.4",
|
||||
authProfileId: "openai-codex:default",
|
||||
}),
|
||||
).rejects.toThrow("must not be a symlink");
|
||||
});
|
||||
});
|
||||
81
openclaw/extensions/openai/openai-codex-cli-bridge.ts
Normal file
81
openclaw/extensions/openai/openai-codex-cli-bridge.ts
Normal file
|
|
@ -0,0 +1,81 @@
|
|||
import crypto from "node:crypto";
|
||||
import path from "node:path";
|
||||
import type {
|
||||
CliBackendPreparedExecution,
|
||||
CliBackendPrepareExecutionContext,
|
||||
} from "openclaw/plugin-sdk/cli-backend";
|
||||
import {
|
||||
ensureAuthProfileStoreForLocalUpdate,
|
||||
type OAuthCredential,
|
||||
} from "openclaw/plugin-sdk/provider-auth";
|
||||
import { writePrivateSecretFileAtomic } from "openclaw/plugin-sdk/secret-file-runtime";
|
||||
|
||||
const OPENAI_CODEX_PROVIDER_ID = "openai-codex";
|
||||
const CODEX_AUTH_ENV_CLEAR_KEYS = ["OPENAI_API_KEY"] as const;
|
||||
|
||||
function isCodexBridgeableOAuthCredential(value: unknown): value is OAuthCredential {
|
||||
return Boolean(
|
||||
value &&
|
||||
typeof value === "object" &&
|
||||
value !== null &&
|
||||
"type" in value &&
|
||||
"provider" in value &&
|
||||
"access" in value &&
|
||||
"refresh" in value &&
|
||||
value.type === "oauth" &&
|
||||
value.provider === OPENAI_CODEX_PROVIDER_ID &&
|
||||
typeof value.access === "string" &&
|
||||
value.access.trim().length > 0 &&
|
||||
typeof value.refresh === "string" &&
|
||||
value.refresh.trim().length > 0,
|
||||
);
|
||||
}
|
||||
|
||||
function resolveCodexBridgeHome(agentDir: string, profileId: string): string {
|
||||
const digest = crypto.createHash("sha256").update(profileId).digest("hex").slice(0, 16);
|
||||
return path.join(agentDir, "cli-auth", "codex", digest);
|
||||
}
|
||||
|
||||
function buildCodexAuthFile(credential: OAuthCredential): string {
|
||||
return `${JSON.stringify(
|
||||
{
|
||||
auth_mode: "chatgpt",
|
||||
tokens: {
|
||||
...(credential.idToken ? { id_token: credential.idToken } : {}),
|
||||
access_token: credential.access,
|
||||
refresh_token: credential.refresh,
|
||||
...(credential.accountId ? { account_id: credential.accountId } : {}),
|
||||
},
|
||||
},
|
||||
null,
|
||||
2,
|
||||
)}\n`;
|
||||
}
|
||||
|
||||
export async function prepareOpenAICodexCliExecution(
|
||||
ctx: CliBackendPrepareExecutionContext,
|
||||
): Promise<CliBackendPreparedExecution | null> {
|
||||
if (!ctx.agentDir || !ctx.authProfileId) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const store = ensureAuthProfileStoreForLocalUpdate(ctx.agentDir);
|
||||
const credential = store.profiles[ctx.authProfileId];
|
||||
if (!isCodexBridgeableOAuthCredential(credential)) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const codexHome = resolveCodexBridgeHome(ctx.agentDir, ctx.authProfileId);
|
||||
await writePrivateSecretFileAtomic({
|
||||
rootDir: ctx.agentDir,
|
||||
filePath: path.join(codexHome, "auth.json"),
|
||||
content: buildCodexAuthFile(credential),
|
||||
});
|
||||
|
||||
return {
|
||||
env: {
|
||||
CODEX_HOME: codexHome,
|
||||
},
|
||||
clearEnv: [...CODEX_AUTH_ENV_CLEAR_KEYS],
|
||||
};
|
||||
}
|
||||
19
openclaw/extensions/openai/openai-codex-provider.runtime.ts
Normal file
19
openclaw/extensions/openai/openai-codex-provider.runtime.ts
Normal file
|
|
@ -0,0 +1,19 @@
|
|||
import {
|
||||
getOAuthApiKey as getOAuthApiKeyFromPi,
|
||||
refreshOpenAICodexToken as refreshOpenAICodexTokenFromPi,
|
||||
} from "@mariozechner/pi-ai/oauth";
|
||||
import { ensureGlobalUndiciEnvProxyDispatcher } from "openclaw/plugin-sdk/runtime-env";
|
||||
|
||||
export async function getOAuthApiKey(
|
||||
...args: Parameters<typeof getOAuthApiKeyFromPi>
|
||||
): Promise<Awaited<ReturnType<typeof getOAuthApiKeyFromPi>>> {
|
||||
ensureGlobalUndiciEnvProxyDispatcher();
|
||||
return await getOAuthApiKeyFromPi(...args);
|
||||
}
|
||||
|
||||
export async function refreshOpenAICodexToken(
|
||||
...args: Parameters<typeof refreshOpenAICodexTokenFromPi>
|
||||
): Promise<Awaited<ReturnType<typeof refreshOpenAICodexTokenFromPi>>> {
|
||||
ensureGlobalUndiciEnvProxyDispatcher();
|
||||
return await refreshOpenAICodexTokenFromPi(...args);
|
||||
}
|
||||
558
openclaw/extensions/openai/openai-codex-provider.test.ts
Normal file
558
openclaw/extensions/openai/openai-codex-provider.test.ts
Normal file
|
|
@ -0,0 +1,558 @@
|
|||
import fs from "node:fs/promises";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "vitest";
|
||||
|
||||
const refreshOpenAICodexTokenMock = vi.hoisted(() => vi.fn());
|
||||
const readOpenAICodexCliOAuthProfileMock = vi.hoisted(() => vi.fn());
|
||||
|
||||
vi.mock("./openai-codex-provider.runtime.js", () => ({
|
||||
refreshOpenAICodexToken: refreshOpenAICodexTokenMock,
|
||||
}));
|
||||
|
||||
vi.mock("./openai-codex-cli-auth.js", async (importOriginal) => {
|
||||
const actual = await importOriginal<typeof import("./openai-codex-cli-auth.js")>();
|
||||
return {
|
||||
...actual,
|
||||
readOpenAICodexCliOAuthProfile: readOpenAICodexCliOAuthProfileMock,
|
||||
};
|
||||
});
|
||||
|
||||
let buildOpenAICodexProviderPlugin: typeof import("./openai-codex-provider.js").buildOpenAICodexProviderPlugin;
|
||||
const tempDirs: string[] = [];
|
||||
|
||||
describe("openai codex provider", () => {
|
||||
beforeAll(async () => {
|
||||
({ buildOpenAICodexProviderPlugin } = await import("./openai-codex-provider.js"));
|
||||
});
|
||||
|
||||
beforeEach(() => {
|
||||
refreshOpenAICodexTokenMock.mockReset();
|
||||
readOpenAICodexCliOAuthProfileMock.mockReset();
|
||||
});
|
||||
|
||||
afterEach(async () => {
|
||||
await Promise.all(
|
||||
tempDirs.splice(0).map((dir) => fs.rm(dir, { recursive: true, force: true })),
|
||||
);
|
||||
});
|
||||
|
||||
it("falls back to the cached credential when accountId extraction fails", async () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
const credential = {
|
||||
type: "oauth" as const,
|
||||
provider: "openai-codex",
|
||||
access: "cached-access-token",
|
||||
refresh: "refresh-token",
|
||||
expires: Date.now() - 60_000,
|
||||
};
|
||||
refreshOpenAICodexTokenMock.mockRejectedValueOnce(
|
||||
new Error("Failed to extract accountId from token"),
|
||||
);
|
||||
|
||||
await expect(provider.refreshOAuth?.(credential)).resolves.toEqual(credential);
|
||||
});
|
||||
|
||||
it("rethrows unrelated refresh failures", async () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
const credential = {
|
||||
type: "oauth" as const,
|
||||
provider: "openai-codex",
|
||||
access: "cached-access-token",
|
||||
refresh: "refresh-token",
|
||||
expires: Date.now() - 60_000,
|
||||
};
|
||||
refreshOpenAICodexTokenMock.mockRejectedValueOnce(new Error("invalid_grant"));
|
||||
|
||||
await expect(provider.refreshOAuth?.(credential)).rejects.toThrow("invalid_grant");
|
||||
});
|
||||
|
||||
it("merges refreshed oauth credentials", async () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
const credential = {
|
||||
type: "oauth" as const,
|
||||
provider: "openai-codex",
|
||||
access: "cached-access-token",
|
||||
refresh: "refresh-token",
|
||||
expires: Date.now() - 60_000,
|
||||
email: "user@example.com",
|
||||
displayName: "User",
|
||||
};
|
||||
refreshOpenAICodexTokenMock.mockResolvedValueOnce({
|
||||
access: "next-access",
|
||||
refresh: "next-refresh",
|
||||
expires: Date.now() + 60_000,
|
||||
});
|
||||
|
||||
await expect(provider.refreshOAuth?.(credential)).resolves.toEqual({
|
||||
...credential,
|
||||
access: "next-access",
|
||||
refresh: "next-refresh",
|
||||
expires: expect.any(Number),
|
||||
});
|
||||
});
|
||||
|
||||
it("returns deprecated-profile doctor guidance for legacy Codex CLI ids", () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
|
||||
expect(
|
||||
provider.buildAuthDoctorHint?.({
|
||||
provider: "openai-codex",
|
||||
profileId: "openai-codex:codex-cli",
|
||||
config: undefined,
|
||||
store: { version: 1, profiles: {} },
|
||||
}),
|
||||
).toBe(
|
||||
"Deprecated profile. Run `openclaw models auth login --provider openai-codex` or `openclaw configure`.",
|
||||
);
|
||||
});
|
||||
|
||||
it("offers explicit browser and one-time Codex CLI import auth methods", () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
|
||||
expect(provider.auth?.map((method) => method.id)).toEqual(["oauth", "import-codex-cli"]);
|
||||
expect(provider.auth?.find((method) => method.id === "import-codex-cli")).toMatchObject({
|
||||
label: "Import Codex CLI login",
|
||||
hint: "Use existing .codex auth once",
|
||||
kind: "oauth",
|
||||
});
|
||||
});
|
||||
|
||||
it("exposes Codex CLI auth as a runtime-only external profile", () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
const credential = {
|
||||
type: "oauth" as const,
|
||||
provider: "openai-codex",
|
||||
access: "access-token",
|
||||
refresh: "refresh-token",
|
||||
expires: Date.now() + 60_000,
|
||||
accountId: "acct-123",
|
||||
};
|
||||
readOpenAICodexCliOAuthProfileMock.mockReturnValueOnce({
|
||||
profileId: "openai-codex:default",
|
||||
credential,
|
||||
});
|
||||
|
||||
expect(
|
||||
provider.resolveExternalAuthProfiles?.({
|
||||
env: { CODEX_HOME: "/sandboxed/codex-home" } as NodeJS.ProcessEnv,
|
||||
store: { version: 1, profiles: {} },
|
||||
}),
|
||||
).toEqual([
|
||||
{
|
||||
profileId: "openai-codex:default",
|
||||
credential,
|
||||
persistence: "runtime-only",
|
||||
},
|
||||
]);
|
||||
expect(readOpenAICodexCliOAuthProfileMock).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
env: expect.objectContaining({ CODEX_HOME: "/sandboxed/codex-home" }),
|
||||
store: { version: 1, profiles: {} },
|
||||
}),
|
||||
);
|
||||
});
|
||||
|
||||
it("uses the provider auth context env when importing Codex CLI auth", async () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
const importMethod = provider.auth?.find((method) => method.id === "import-codex-cli");
|
||||
const agentDir = await fs.mkdtemp(path.join(os.tmpdir(), "openclaw-openai-codex-provider-"));
|
||||
tempDirs.push(agentDir);
|
||||
readOpenAICodexCliOAuthProfileMock.mockImplementationOnce(({ env }) => {
|
||||
expect(env).toMatchObject({
|
||||
CODEX_HOME: "/sandboxed/codex-home",
|
||||
});
|
||||
return {
|
||||
profileId: "openai-codex:default",
|
||||
credential: {
|
||||
type: "oauth",
|
||||
provider: "openai-codex",
|
||||
access: "access-token",
|
||||
refresh: "refresh-token",
|
||||
expires: Date.now() + 60_000,
|
||||
email: "codex@example.com",
|
||||
displayName: "Codex User",
|
||||
accountId: "acct-123",
|
||||
},
|
||||
};
|
||||
});
|
||||
|
||||
await expect(
|
||||
importMethod?.run({
|
||||
config: {},
|
||||
env: { CODEX_HOME: "/sandboxed/codex-home" },
|
||||
agentDir,
|
||||
prompter: {} as never,
|
||||
runtime: {} as never,
|
||||
isRemote: false,
|
||||
openUrl: async () => {},
|
||||
oauth: { createVpsAwareHandlers: (() => ({})) as never },
|
||||
}),
|
||||
).resolves.toMatchObject({
|
||||
profiles: [
|
||||
{
|
||||
profileId: "openai-codex:default",
|
||||
credential: expect.objectContaining({
|
||||
provider: "openai-codex",
|
||||
access: "access-token",
|
||||
}),
|
||||
},
|
||||
],
|
||||
});
|
||||
});
|
||||
|
||||
it("owns native reasoning output mode for Codex responses", () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
|
||||
expect(
|
||||
provider.resolveReasoningOutputMode?.({
|
||||
provider: "openai-codex",
|
||||
modelApi: "openai-codex-responses",
|
||||
modelId: "gpt-5.4",
|
||||
} as never),
|
||||
).toBe("native");
|
||||
});
|
||||
|
||||
it("resolves gpt-5.4 with native contextWindow plus default contextTokens cap", () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
|
||||
const model = provider.resolveDynamicModel?.({
|
||||
provider: "openai-codex",
|
||||
modelId: "gpt-5.4",
|
||||
modelRegistry: {
|
||||
find: (providerId: string, modelId: string) => {
|
||||
if (providerId === "openai-codex" && modelId === "gpt-5.3-codex") {
|
||||
return {
|
||||
id: "gpt-5.3-codex",
|
||||
name: "gpt-5.3-codex",
|
||||
provider: "openai-codex",
|
||||
api: "openai-codex-responses",
|
||||
baseUrl: "https://chatgpt.com/backend-api",
|
||||
reasoning: true,
|
||||
input: ["text", "image"] as const,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 272_000,
|
||||
maxTokens: 128_000,
|
||||
};
|
||||
}
|
||||
return undefined;
|
||||
},
|
||||
} as never,
|
||||
});
|
||||
|
||||
expect(model).toMatchObject({
|
||||
id: "gpt-5.4",
|
||||
contextWindow: 1_050_000,
|
||||
contextTokens: 272_000,
|
||||
maxTokens: 128_000,
|
||||
});
|
||||
});
|
||||
|
||||
it("resolves gpt-5.4-pro with pro pricing and codex-sized limits", () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
|
||||
const model = provider.resolveDynamicModel?.({
|
||||
provider: "openai-codex",
|
||||
modelId: "gpt-5.4-pro",
|
||||
modelRegistry: {
|
||||
find: (providerId: string, modelId: string) => {
|
||||
if (providerId === "openai-codex" && modelId === "gpt-5.3-codex") {
|
||||
return {
|
||||
id: "gpt-5.3-codex",
|
||||
name: "gpt-5.3-codex",
|
||||
provider: "openai-codex",
|
||||
api: "openai-codex-responses",
|
||||
baseUrl: "https://chatgpt.com/backend-api",
|
||||
reasoning: true,
|
||||
input: ["text", "image"] as const,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 272_000,
|
||||
maxTokens: 128_000,
|
||||
};
|
||||
}
|
||||
return undefined;
|
||||
},
|
||||
} as never,
|
||||
});
|
||||
|
||||
expect(model).toMatchObject({
|
||||
id: "gpt-5.4-pro",
|
||||
contextWindow: 1_050_000,
|
||||
contextTokens: 272_000,
|
||||
maxTokens: 128_000,
|
||||
cost: { input: 30, output: 180, cacheRead: 0, cacheWrite: 0 },
|
||||
});
|
||||
});
|
||||
|
||||
it("resolves gpt-5.4-pro from a gpt-5.4 runtime template when legacy codex rows are absent", () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
|
||||
const model = provider.resolveDynamicModel?.({
|
||||
provider: "openai-codex",
|
||||
modelId: "gpt-5.4-pro",
|
||||
modelRegistry: {
|
||||
find: (providerId: string, modelId: string) => {
|
||||
if (providerId === "openai-codex" && modelId === "gpt-5.4") {
|
||||
return {
|
||||
id: "gpt-5.4",
|
||||
name: "gpt-5.4",
|
||||
provider: "openai-codex",
|
||||
api: "openai-codex-responses",
|
||||
baseUrl: "https://chatgpt.com/backend-api",
|
||||
reasoning: true,
|
||||
input: ["text", "image"] as const,
|
||||
cost: { input: 2.5, output: 15, cacheRead: 0.25, cacheWrite: 0 },
|
||||
contextWindow: 1_050_000,
|
||||
contextTokens: 272_000,
|
||||
maxTokens: 128_000,
|
||||
};
|
||||
}
|
||||
return undefined;
|
||||
},
|
||||
} as never,
|
||||
});
|
||||
|
||||
expect(model).toMatchObject({
|
||||
id: "gpt-5.4-pro",
|
||||
api: "openai-codex-responses",
|
||||
baseUrl: "https://chatgpt.com/backend-api",
|
||||
contextWindow: 1_050_000,
|
||||
contextTokens: 272_000,
|
||||
maxTokens: 128_000,
|
||||
cost: { input: 30, output: 180, cacheRead: 0, cacheWrite: 0 },
|
||||
});
|
||||
});
|
||||
|
||||
it("resolves the legacy gpt-5.4-codex alias to canonical gpt-5.4", () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
|
||||
const model = provider.resolveDynamicModel?.({
|
||||
provider: "openai-codex",
|
||||
modelId: "gpt-5.4-codex",
|
||||
modelRegistry: {
|
||||
find: (providerId: string, modelId: string) => {
|
||||
if (providerId === "openai-codex" && modelId === "gpt-5.3-codex") {
|
||||
return {
|
||||
id: "gpt-5.3-codex",
|
||||
name: "gpt-5.3-codex",
|
||||
provider: "openai-codex",
|
||||
api: "openai-codex-responses",
|
||||
baseUrl: "https://chatgpt.com/backend-api",
|
||||
reasoning: true,
|
||||
input: ["text", "image"] as const,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 272_000,
|
||||
maxTokens: 128_000,
|
||||
};
|
||||
}
|
||||
return undefined;
|
||||
},
|
||||
} as never,
|
||||
});
|
||||
|
||||
expect(model).toMatchObject({
|
||||
id: "gpt-5.4",
|
||||
name: "gpt-5.4",
|
||||
contextWindow: 1_050_000,
|
||||
contextTokens: 272_000,
|
||||
maxTokens: 128_000,
|
||||
});
|
||||
});
|
||||
|
||||
it("resolves gpt-5.4-mini from codex templates with codex-sized limits", () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
|
||||
const model = provider.resolveDynamicModel?.({
|
||||
provider: "openai-codex",
|
||||
modelId: "gpt-5.4-mini",
|
||||
modelRegistry: {
|
||||
find: (providerId: string, modelId: string) => {
|
||||
if (providerId === "openai-codex" && modelId === "gpt-5.1-codex-mini") {
|
||||
return {
|
||||
id: "gpt-5.1-codex-mini",
|
||||
name: "gpt-5.1-codex-mini",
|
||||
provider: "openai-codex",
|
||||
api: "openai-codex-responses",
|
||||
baseUrl: "https://chatgpt.com/backend-api",
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
cost: { input: 0.25, output: 2, cacheRead: 0.025, cacheWrite: 0 },
|
||||
contextWindow: 272_000,
|
||||
maxTokens: 128_000,
|
||||
};
|
||||
}
|
||||
return null;
|
||||
},
|
||||
} as never,
|
||||
} as never);
|
||||
|
||||
expect(model).toMatchObject({
|
||||
id: "gpt-5.4-mini",
|
||||
contextWindow: 272_000,
|
||||
maxTokens: 128_000,
|
||||
cost: { input: 0.75, output: 4.5, cacheRead: 0.075, cacheWrite: 0 },
|
||||
});
|
||||
expect(model).not.toHaveProperty("contextTokens");
|
||||
});
|
||||
|
||||
it("augments catalog with gpt-5.4 native contextWindow and runtime cap", () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
|
||||
const entries = provider.augmentModelCatalog?.({
|
||||
env: process.env,
|
||||
entries: [
|
||||
{
|
||||
id: "gpt-5.3-codex",
|
||||
name: "gpt-5.3-codex",
|
||||
provider: "openai-codex",
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
contextWindow: 272_000,
|
||||
},
|
||||
],
|
||||
} as never);
|
||||
|
||||
expect(entries).toContainEqual(
|
||||
expect.objectContaining({
|
||||
id: "gpt-5.4",
|
||||
contextWindow: 1_050_000,
|
||||
contextTokens: 272_000,
|
||||
cost: { input: 2.5, output: 15, cacheRead: 0.25, cacheWrite: 0 },
|
||||
}),
|
||||
);
|
||||
expect(entries).toContainEqual(
|
||||
expect.objectContaining({
|
||||
id: "gpt-5.4-pro",
|
||||
contextWindow: 1_050_000,
|
||||
contextTokens: 272_000,
|
||||
cost: { input: 30, output: 180, cacheRead: 0, cacheWrite: 0 },
|
||||
}),
|
||||
);
|
||||
expect(entries).toContainEqual(
|
||||
expect.objectContaining({
|
||||
id: "gpt-5.4-mini",
|
||||
contextWindow: 272_000,
|
||||
cost: { input: 0.75, output: 4.5, cacheRead: 0.075, cacheWrite: 0 },
|
||||
}),
|
||||
);
|
||||
});
|
||||
|
||||
it("augments gpt-5.4-pro from catalog gpt-5.4 when legacy codex rows are absent", () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
|
||||
const entries = provider.augmentModelCatalog?.({
|
||||
env: process.env,
|
||||
entries: [
|
||||
{
|
||||
id: "gpt-5.4",
|
||||
name: "gpt-5.4",
|
||||
provider: "openai-codex",
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
contextWindow: 272_000,
|
||||
},
|
||||
],
|
||||
} as never);
|
||||
|
||||
expect(entries).toContainEqual(
|
||||
expect.objectContaining({
|
||||
id: "gpt-5.4-pro",
|
||||
contextWindow: 1_050_000,
|
||||
contextTokens: 272_000,
|
||||
cost: { input: 30, output: 180, cacheRead: 0, cacheWrite: 0 },
|
||||
}),
|
||||
);
|
||||
});
|
||||
|
||||
it("canonicalizes legacy gpt-5.4-codex models during resolved-model normalization", () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
|
||||
const model = provider.normalizeResolvedModel?.({
|
||||
provider: "openai-codex",
|
||||
model: {
|
||||
id: "gpt-5.4-codex",
|
||||
name: "gpt-5.4-codex",
|
||||
provider: "openai-codex",
|
||||
api: "openai-codex-responses",
|
||||
baseUrl: "https://chatgpt.com/backend-api",
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 1_050_000,
|
||||
contextTokens: 272_000,
|
||||
maxTokens: 128_000,
|
||||
},
|
||||
} as never);
|
||||
|
||||
expect(model).toMatchObject({
|
||||
id: "gpt-5.4",
|
||||
name: "gpt-5.4",
|
||||
});
|
||||
});
|
||||
|
||||
it("defaults missing codex api metadata to openai-codex-responses", () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
|
||||
const model = provider.normalizeResolvedModel?.({
|
||||
provider: "openai-codex",
|
||||
model: {
|
||||
id: "gpt-5.4",
|
||||
name: "gpt-5.4",
|
||||
provider: "openai-codex",
|
||||
baseUrl: "https://chatgpt.com/backend-api",
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 1_050_000,
|
||||
contextTokens: 272_000,
|
||||
maxTokens: 128_000,
|
||||
},
|
||||
} as never);
|
||||
|
||||
expect(model).toMatchObject({
|
||||
api: "openai-codex-responses",
|
||||
baseUrl: "https://chatgpt.com/backend-api",
|
||||
});
|
||||
});
|
||||
|
||||
it("normalizes stale /backend-api/v1 codex metadata to the canonical base url", () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
|
||||
const model = provider.normalizeResolvedModel?.({
|
||||
provider: "openai-codex",
|
||||
model: {
|
||||
id: "gpt-5.4",
|
||||
name: "gpt-5.4",
|
||||
provider: "openai-codex",
|
||||
api: "openai-codex-responses",
|
||||
baseUrl: "https://chatgpt.com/backend-api/v1",
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 1_050_000,
|
||||
contextTokens: 272_000,
|
||||
maxTokens: 128_000,
|
||||
},
|
||||
} as never);
|
||||
|
||||
expect(model).toMatchObject({
|
||||
api: "openai-codex-responses",
|
||||
baseUrl: "https://chatgpt.com/backend-api",
|
||||
});
|
||||
});
|
||||
|
||||
it("normalizes transport metadata for stale /backend-api/v1 codex routes", () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
|
||||
expect(
|
||||
provider.normalizeTransport?.({
|
||||
provider: "openai-codex",
|
||||
api: "openai-codex-responses",
|
||||
baseUrl: "https://chatgpt.com/backend-api/v1",
|
||||
} as never),
|
||||
).toEqual({
|
||||
api: "openai-codex-responses",
|
||||
baseUrl: "https://chatgpt.com/backend-api",
|
||||
});
|
||||
});
|
||||
});
|
||||
469
openclaw/extensions/openai/openai-codex-provider.ts
Normal file
469
openclaw/extensions/openai/openai-codex-provider.ts
Normal file
|
|
@ -0,0 +1,469 @@
|
|||
import { formatErrorMessage } from "openclaw/plugin-sdk/error-runtime";
|
||||
import type {
|
||||
ProviderAuthContext,
|
||||
ProviderResolveDynamicModelContext,
|
||||
ProviderRuntimeModel,
|
||||
} from "openclaw/plugin-sdk/plugin-entry";
|
||||
import {
|
||||
ensureAuthProfileStoreForLocalUpdate,
|
||||
listProfilesForProvider,
|
||||
type OAuthCredential,
|
||||
type ProviderAuthResult,
|
||||
} from "openclaw/plugin-sdk/provider-auth";
|
||||
import { buildOauthProviderAuthResult } from "openclaw/plugin-sdk/provider-auth";
|
||||
import { loginOpenAICodexOAuth } from "openclaw/plugin-sdk/provider-auth-login";
|
||||
import {
|
||||
DEFAULT_CONTEXT_TOKENS,
|
||||
normalizeModelCompat,
|
||||
normalizeProviderId,
|
||||
type ProviderPlugin,
|
||||
} from "openclaw/plugin-sdk/provider-model-shared";
|
||||
import { fetchCodexUsage } from "openclaw/plugin-sdk/provider-usage";
|
||||
import { normalizeLowercaseStringOrEmpty, readStringValue } from "openclaw/plugin-sdk/text-runtime";
|
||||
import { isOpenAIApiBaseUrl, isOpenAICodexBaseUrl } from "./base-url.js";
|
||||
import { OPENAI_CODEX_DEFAULT_MODEL } from "./default-models.js";
|
||||
import { resolveCodexAuthIdentity } from "./openai-codex-auth-identity.js";
|
||||
import { buildOpenAICodexProvider } from "./openai-codex-catalog.js";
|
||||
import { CODEX_CLI_PROFILE_ID, readOpenAICodexCliOAuthProfile } from "./openai-codex-cli-auth.js";
|
||||
import {
|
||||
buildOpenAIResponsesProviderHooks,
|
||||
buildOpenAISyntheticCatalogEntry,
|
||||
cloneFirstTemplateModel,
|
||||
findCatalogTemplate,
|
||||
matchesExactOrPrefix,
|
||||
} from "./shared.js";
|
||||
|
||||
const PROVIDER_ID = "openai-codex";
|
||||
const OPENAI_CODEX_BASE_URL = "https://chatgpt.com/backend-api";
|
||||
const OPENAI_CODEX_GPT_54_MODEL_ID = "gpt-5.4";
|
||||
const OPENAI_CODEX_GPT_54_LEGACY_MODEL_ID = "gpt-5.4-codex";
|
||||
const OPENAI_CODEX_GPT_54_PRO_MODEL_ID = "gpt-5.4-pro";
|
||||
const OPENAI_CODEX_GPT_54_MINI_MODEL_ID = "gpt-5.4-mini";
|
||||
const OPENAI_CODEX_GPT_54_NATIVE_CONTEXT_TOKENS = 1_050_000;
|
||||
const OPENAI_CODEX_GPT_54_DEFAULT_CONTEXT_TOKENS = 272_000;
|
||||
const OPENAI_CODEX_GPT_54_MINI_CONTEXT_TOKENS = 272_000;
|
||||
const OPENAI_CODEX_GPT_54_MAX_TOKENS = 128_000;
|
||||
const OPENAI_CODEX_GPT_54_COST = {
|
||||
input: 2.5,
|
||||
output: 15,
|
||||
cacheRead: 0.25,
|
||||
cacheWrite: 0,
|
||||
} as const;
|
||||
const OPENAI_CODEX_GPT_54_PRO_COST = {
|
||||
input: 30,
|
||||
output: 180,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
} as const;
|
||||
const OPENAI_CODEX_GPT_54_MINI_COST = {
|
||||
input: 0.75,
|
||||
output: 4.5,
|
||||
cacheRead: 0.075,
|
||||
cacheWrite: 0,
|
||||
} as const;
|
||||
const OPENAI_CODEX_GPT_54_TEMPLATE_MODEL_IDS = ["gpt-5.3-codex", "gpt-5.2-codex"] as const;
|
||||
/** Legacy codex rows first; fall back to catalog `gpt-5.4` when the API omits 5.3/5.2. */
|
||||
const OPENAI_CODEX_GPT_54_CATALOG_SYNTH_TEMPLATE_MODEL_IDS = [
|
||||
...OPENAI_CODEX_GPT_54_TEMPLATE_MODEL_IDS,
|
||||
OPENAI_CODEX_GPT_54_MODEL_ID,
|
||||
] as const;
|
||||
const OPENAI_CODEX_GPT_54_MINI_TEMPLATE_MODEL_IDS = [
|
||||
OPENAI_CODEX_GPT_54_MODEL_ID,
|
||||
"gpt-5.1-codex-mini",
|
||||
...OPENAI_CODEX_GPT_54_TEMPLATE_MODEL_IDS,
|
||||
] as const;
|
||||
const OPENAI_CODEX_GPT_53_MODEL_ID = "gpt-5.3-codex";
|
||||
const OPENAI_CODEX_GPT_53_SPARK_MODEL_ID = "gpt-5.3-codex-spark";
|
||||
const OPENAI_CODEX_GPT_53_SPARK_CONTEXT_TOKENS = 128_000;
|
||||
const OPENAI_CODEX_GPT_53_SPARK_MAX_TOKENS = 128_000;
|
||||
const OPENAI_CODEX_TEMPLATE_MODEL_IDS = ["gpt-5.2-codex"] as const;
|
||||
const OPENAI_CODEX_XHIGH_MODEL_IDS = [
|
||||
OPENAI_CODEX_GPT_54_MODEL_ID,
|
||||
OPENAI_CODEX_GPT_54_PRO_MODEL_ID,
|
||||
OPENAI_CODEX_GPT_54_MINI_MODEL_ID,
|
||||
OPENAI_CODEX_GPT_53_MODEL_ID,
|
||||
OPENAI_CODEX_GPT_53_SPARK_MODEL_ID,
|
||||
"gpt-5.2-codex",
|
||||
"gpt-5.1-codex",
|
||||
] as const;
|
||||
const OPENAI_CODEX_MODERN_MODEL_IDS = [
|
||||
OPENAI_CODEX_GPT_54_MODEL_ID,
|
||||
OPENAI_CODEX_GPT_54_PRO_MODEL_ID,
|
||||
OPENAI_CODEX_GPT_54_MINI_MODEL_ID,
|
||||
"gpt-5.2",
|
||||
"gpt-5.2-codex",
|
||||
OPENAI_CODEX_GPT_53_MODEL_ID,
|
||||
OPENAI_CODEX_GPT_53_SPARK_MODEL_ID,
|
||||
] as const;
|
||||
|
||||
function normalizeCodexTransportFields(params: {
|
||||
api?: ProviderRuntimeModel["api"] | null;
|
||||
baseUrl?: string;
|
||||
}): {
|
||||
api?: ProviderRuntimeModel["api"];
|
||||
baseUrl?: string;
|
||||
} {
|
||||
const useCodexTransport =
|
||||
!params.baseUrl || isOpenAIApiBaseUrl(params.baseUrl) || isOpenAICodexBaseUrl(params.baseUrl);
|
||||
const api =
|
||||
useCodexTransport && (!params.api || params.api === "openai-responses")
|
||||
? "openai-codex-responses"
|
||||
: (params.api ?? undefined);
|
||||
const baseUrl =
|
||||
api === "openai-codex-responses" && useCodexTransport ? OPENAI_CODEX_BASE_URL : params.baseUrl;
|
||||
return { api, baseUrl };
|
||||
}
|
||||
|
||||
function normalizeCodexTransport(model: ProviderRuntimeModel): ProviderRuntimeModel {
|
||||
const lowerModelId = normalizeLowercaseStringOrEmpty(model.id);
|
||||
const canonicalModelId =
|
||||
lowerModelId === OPENAI_CODEX_GPT_54_LEGACY_MODEL_ID ? OPENAI_CODEX_GPT_54_MODEL_ID : model.id;
|
||||
const canonicalName =
|
||||
normalizeLowercaseStringOrEmpty(model.name) === OPENAI_CODEX_GPT_54_LEGACY_MODEL_ID
|
||||
? OPENAI_CODEX_GPT_54_MODEL_ID
|
||||
: model.name;
|
||||
const normalizedTransport = normalizeCodexTransportFields({
|
||||
api: model.api,
|
||||
baseUrl: model.baseUrl,
|
||||
});
|
||||
const api = normalizedTransport.api ?? model.api;
|
||||
const baseUrl = normalizedTransport.baseUrl ?? model.baseUrl;
|
||||
if (
|
||||
api === model.api &&
|
||||
baseUrl === model.baseUrl &&
|
||||
canonicalModelId === model.id &&
|
||||
canonicalName === model.name
|
||||
) {
|
||||
return model;
|
||||
}
|
||||
return {
|
||||
...model,
|
||||
id: canonicalModelId,
|
||||
name: canonicalName,
|
||||
api,
|
||||
baseUrl,
|
||||
};
|
||||
}
|
||||
|
||||
function resolveCodexForwardCompatModel(ctx: ProviderResolveDynamicModelContext) {
|
||||
const trimmedModelId = ctx.modelId.trim();
|
||||
const lower = normalizeLowercaseStringOrEmpty(trimmedModelId);
|
||||
|
||||
let templateIds: readonly string[];
|
||||
let patch: Parameters<typeof cloneFirstTemplateModel>[0]["patch"];
|
||||
if (lower === OPENAI_CODEX_GPT_54_MODEL_ID || lower === OPENAI_CODEX_GPT_54_LEGACY_MODEL_ID) {
|
||||
templateIds = OPENAI_CODEX_GPT_54_CATALOG_SYNTH_TEMPLATE_MODEL_IDS;
|
||||
patch = {
|
||||
contextWindow: OPENAI_CODEX_GPT_54_NATIVE_CONTEXT_TOKENS,
|
||||
contextTokens: OPENAI_CODEX_GPT_54_DEFAULT_CONTEXT_TOKENS,
|
||||
maxTokens: OPENAI_CODEX_GPT_54_MAX_TOKENS,
|
||||
cost: OPENAI_CODEX_GPT_54_COST,
|
||||
};
|
||||
} else if (lower === OPENAI_CODEX_GPT_54_PRO_MODEL_ID) {
|
||||
templateIds = OPENAI_CODEX_GPT_54_CATALOG_SYNTH_TEMPLATE_MODEL_IDS;
|
||||
patch = {
|
||||
contextWindow: OPENAI_CODEX_GPT_54_NATIVE_CONTEXT_TOKENS,
|
||||
contextTokens: OPENAI_CODEX_GPT_54_DEFAULT_CONTEXT_TOKENS,
|
||||
maxTokens: OPENAI_CODEX_GPT_54_MAX_TOKENS,
|
||||
cost: OPENAI_CODEX_GPT_54_PRO_COST,
|
||||
};
|
||||
} else if (lower === OPENAI_CODEX_GPT_54_MINI_MODEL_ID) {
|
||||
templateIds = OPENAI_CODEX_GPT_54_MINI_TEMPLATE_MODEL_IDS;
|
||||
patch = {
|
||||
contextWindow: OPENAI_CODEX_GPT_54_MINI_CONTEXT_TOKENS,
|
||||
maxTokens: OPENAI_CODEX_GPT_54_MAX_TOKENS,
|
||||
cost: OPENAI_CODEX_GPT_54_MINI_COST,
|
||||
};
|
||||
} else if (lower === OPENAI_CODEX_GPT_53_SPARK_MODEL_ID) {
|
||||
templateIds = [OPENAI_CODEX_GPT_53_MODEL_ID, ...OPENAI_CODEX_TEMPLATE_MODEL_IDS];
|
||||
patch = {
|
||||
api: "openai-codex-responses",
|
||||
provider: PROVIDER_ID,
|
||||
baseUrl: OPENAI_CODEX_BASE_URL,
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: OPENAI_CODEX_GPT_53_SPARK_CONTEXT_TOKENS,
|
||||
maxTokens: OPENAI_CODEX_GPT_53_SPARK_MAX_TOKENS,
|
||||
};
|
||||
} else if (lower === OPENAI_CODEX_GPT_53_MODEL_ID) {
|
||||
templateIds = OPENAI_CODEX_TEMPLATE_MODEL_IDS;
|
||||
} else {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
return (
|
||||
cloneFirstTemplateModel({
|
||||
providerId: PROVIDER_ID,
|
||||
modelId:
|
||||
lower === OPENAI_CODEX_GPT_54_LEGACY_MODEL_ID
|
||||
? OPENAI_CODEX_GPT_54_MODEL_ID
|
||||
: trimmedModelId,
|
||||
templateIds,
|
||||
ctx,
|
||||
patch,
|
||||
}) ??
|
||||
normalizeModelCompat({
|
||||
id:
|
||||
lower === OPENAI_CODEX_GPT_54_LEGACY_MODEL_ID
|
||||
? OPENAI_CODEX_GPT_54_MODEL_ID
|
||||
: trimmedModelId,
|
||||
name:
|
||||
lower === OPENAI_CODEX_GPT_54_LEGACY_MODEL_ID
|
||||
? OPENAI_CODEX_GPT_54_MODEL_ID
|
||||
: trimmedModelId,
|
||||
api: "openai-codex-responses",
|
||||
provider: PROVIDER_ID,
|
||||
baseUrl: OPENAI_CODEX_BASE_URL,
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: patch?.contextWindow ?? DEFAULT_CONTEXT_TOKENS,
|
||||
contextTokens: patch?.contextTokens,
|
||||
maxTokens: patch?.maxTokens ?? DEFAULT_CONTEXT_TOKENS,
|
||||
} as ProviderRuntimeModel)
|
||||
);
|
||||
}
|
||||
|
||||
async function refreshOpenAICodexOAuthCredential(cred: OAuthCredential) {
|
||||
try {
|
||||
const { refreshOpenAICodexToken } = await import("./openai-codex-provider.runtime.js");
|
||||
const refreshed = await refreshOpenAICodexToken(cred.refresh);
|
||||
return {
|
||||
...cred,
|
||||
...refreshed,
|
||||
type: "oauth" as const,
|
||||
provider: PROVIDER_ID,
|
||||
email: cred.email,
|
||||
displayName: cred.displayName,
|
||||
};
|
||||
} catch (error) {
|
||||
const message = formatErrorMessage(error);
|
||||
if (
|
||||
/extract\s+accountid\s+from\s+token/i.test(message) &&
|
||||
typeof cred.access === "string" &&
|
||||
cred.access.trim().length > 0
|
||||
) {
|
||||
return cred;
|
||||
}
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
async function runOpenAICodexOAuth(ctx: ProviderAuthContext) {
|
||||
let creds;
|
||||
try {
|
||||
creds = await loginOpenAICodexOAuth({
|
||||
prompter: ctx.prompter,
|
||||
runtime: ctx.runtime,
|
||||
isRemote: ctx.isRemote,
|
||||
openUrl: ctx.openUrl,
|
||||
localBrowserMessage: "Complete sign-in in browser…",
|
||||
});
|
||||
} catch {
|
||||
return { profiles: [] };
|
||||
}
|
||||
if (!creds) {
|
||||
return { profiles: [] };
|
||||
}
|
||||
|
||||
const identity = resolveCodexAuthIdentity({
|
||||
accessToken: creds.access,
|
||||
email: readStringValue(creds.email),
|
||||
});
|
||||
|
||||
return buildOauthProviderAuthResult({
|
||||
providerId: PROVIDER_ID,
|
||||
defaultModel: OPENAI_CODEX_DEFAULT_MODEL,
|
||||
access: creds.access,
|
||||
refresh: creds.refresh,
|
||||
expires: creds.expires,
|
||||
email: identity.email,
|
||||
profileName: identity.profileName,
|
||||
});
|
||||
}
|
||||
|
||||
async function runImportOpenAICodexCliAuth(ctx: ProviderAuthContext) {
|
||||
const profile = readOpenAICodexCliOAuthProfile({
|
||||
env: ctx.env ?? process.env,
|
||||
store: ensureAuthProfileStoreForLocalUpdate(ctx.agentDir),
|
||||
});
|
||||
if (!profile) {
|
||||
throw new Error(
|
||||
"No compatible Codex CLI OAuth login found. Sign in with `codex` first or use ChatGPT OAuth instead.",
|
||||
);
|
||||
}
|
||||
|
||||
return {
|
||||
profiles: [{ profileId: profile.profileId, credential: profile.credential }],
|
||||
configPatch: {
|
||||
agents: {
|
||||
defaults: {
|
||||
models: {
|
||||
[OPENAI_CODEX_DEFAULT_MODEL]: {},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
defaultModel: OPENAI_CODEX_DEFAULT_MODEL,
|
||||
notes: ["Imported existing Codex CLI login into OpenClaw canonical auth."],
|
||||
} satisfies ProviderAuthResult;
|
||||
}
|
||||
|
||||
function ensureOpenAICodexCatalogAuthStore(ctx: { agentDir?: string; env?: NodeJS.ProcessEnv }) {
|
||||
const store = ensureAuthProfileStoreForLocalUpdate(ctx.agentDir);
|
||||
const profile = readOpenAICodexCliOAuthProfile({
|
||||
env: ctx.env ?? process.env,
|
||||
store,
|
||||
});
|
||||
if (!profile) {
|
||||
return store;
|
||||
}
|
||||
return {
|
||||
...store,
|
||||
profiles: {
|
||||
...store.profiles,
|
||||
[profile.profileId]: profile.credential,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
function buildOpenAICodexAuthDoctorHint(ctx: { profileId?: string }) {
|
||||
if (ctx.profileId !== CODEX_CLI_PROFILE_ID) {
|
||||
return undefined;
|
||||
}
|
||||
return "Deprecated profile. Run `openclaw models auth login --provider openai-codex` or `openclaw configure`.";
|
||||
}
|
||||
|
||||
export function buildOpenAICodexProviderPlugin(): ProviderPlugin {
|
||||
return {
|
||||
id: PROVIDER_ID,
|
||||
label: "OpenAI Codex",
|
||||
docsPath: "/providers/models",
|
||||
auth: [
|
||||
{
|
||||
id: "oauth",
|
||||
label: "ChatGPT OAuth",
|
||||
hint: "Browser sign-in",
|
||||
kind: "oauth",
|
||||
run: async (ctx) => await runOpenAICodexOAuth(ctx),
|
||||
},
|
||||
{
|
||||
id: "import-codex-cli",
|
||||
label: "Import Codex CLI login",
|
||||
hint: "Use existing .codex auth once",
|
||||
kind: "oauth",
|
||||
run: async (ctx) => await runImportOpenAICodexCliAuth(ctx),
|
||||
},
|
||||
],
|
||||
wizard: {
|
||||
setup: {
|
||||
choiceId: "openai-codex",
|
||||
choiceLabel: "OpenAI Codex (ChatGPT OAuth)",
|
||||
choiceHint: "Browser sign-in",
|
||||
methodId: "oauth",
|
||||
},
|
||||
},
|
||||
catalog: {
|
||||
order: "profile",
|
||||
run: async (ctx) => {
|
||||
const authStore = ensureOpenAICodexCatalogAuthStore(ctx);
|
||||
if (listProfilesForProvider(authStore, PROVIDER_ID).length === 0) {
|
||||
return null;
|
||||
}
|
||||
return {
|
||||
provider: buildOpenAICodexProvider(),
|
||||
};
|
||||
},
|
||||
},
|
||||
resolveDynamicModel: (ctx) => resolveCodexForwardCompatModel(ctx),
|
||||
buildAuthDoctorHint: (ctx) => buildOpenAICodexAuthDoctorHint(ctx),
|
||||
supportsXHighThinking: ({ modelId }) =>
|
||||
matchesExactOrPrefix(modelId, OPENAI_CODEX_XHIGH_MODEL_IDS),
|
||||
isModernModelRef: ({ modelId }) => matchesExactOrPrefix(modelId, OPENAI_CODEX_MODERN_MODEL_IDS),
|
||||
preferRuntimeResolvedModel: (ctx) => {
|
||||
if (normalizeProviderId(ctx.provider) !== PROVIDER_ID) {
|
||||
return false;
|
||||
}
|
||||
const id = ctx.modelId.trim().toLowerCase();
|
||||
return id === OPENAI_CODEX_GPT_54_MODEL_ID || id === OPENAI_CODEX_GPT_54_PRO_MODEL_ID;
|
||||
},
|
||||
...buildOpenAIResponsesProviderHooks(),
|
||||
resolveReasoningOutputMode: () => "native",
|
||||
normalizeResolvedModel: (ctx) => {
|
||||
if (normalizeProviderId(ctx.provider) !== PROVIDER_ID) {
|
||||
return undefined;
|
||||
}
|
||||
return normalizeCodexTransport(ctx.model);
|
||||
},
|
||||
normalizeTransport: ({ provider, api, baseUrl }) => {
|
||||
if (normalizeProviderId(provider) !== PROVIDER_ID) {
|
||||
return undefined;
|
||||
}
|
||||
const normalized = normalizeCodexTransportFields({ api, baseUrl });
|
||||
if (normalized.api === api && normalized.baseUrl === baseUrl) {
|
||||
return undefined;
|
||||
}
|
||||
return normalized;
|
||||
},
|
||||
resolveUsageAuth: async (ctx) => await ctx.resolveOAuthToken(),
|
||||
fetchUsageSnapshot: async (ctx) =>
|
||||
await fetchCodexUsage(ctx.token, ctx.accountId, ctx.timeoutMs, ctx.fetchFn),
|
||||
refreshOAuth: async (cred) => await refreshOpenAICodexOAuthCredential(cred),
|
||||
resolveExternalAuthProfiles: (ctx) => {
|
||||
const profile = readOpenAICodexCliOAuthProfile({
|
||||
env: ctx.env,
|
||||
store: ctx.store,
|
||||
});
|
||||
return profile ? [{ ...profile, persistence: "runtime-only" }] : [];
|
||||
},
|
||||
augmentModelCatalog: (ctx) => {
|
||||
const gpt54Template = findCatalogTemplate({
|
||||
entries: ctx.entries,
|
||||
providerId: PROVIDER_ID,
|
||||
templateIds: OPENAI_CODEX_GPT_54_CATALOG_SYNTH_TEMPLATE_MODEL_IDS,
|
||||
});
|
||||
const gpt54MiniTemplate = findCatalogTemplate({
|
||||
entries: ctx.entries,
|
||||
providerId: PROVIDER_ID,
|
||||
templateIds: OPENAI_CODEX_GPT_54_MINI_TEMPLATE_MODEL_IDS,
|
||||
});
|
||||
const sparkTemplate = findCatalogTemplate({
|
||||
entries: ctx.entries,
|
||||
providerId: PROVIDER_ID,
|
||||
templateIds: [OPENAI_CODEX_GPT_53_MODEL_ID, ...OPENAI_CODEX_TEMPLATE_MODEL_IDS],
|
||||
});
|
||||
return [
|
||||
buildOpenAISyntheticCatalogEntry(gpt54Template, {
|
||||
id: OPENAI_CODEX_GPT_54_MODEL_ID,
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
contextWindow: OPENAI_CODEX_GPT_54_NATIVE_CONTEXT_TOKENS,
|
||||
contextTokens: OPENAI_CODEX_GPT_54_DEFAULT_CONTEXT_TOKENS,
|
||||
cost: OPENAI_CODEX_GPT_54_COST,
|
||||
}),
|
||||
buildOpenAISyntheticCatalogEntry(gpt54Template, {
|
||||
id: OPENAI_CODEX_GPT_54_PRO_MODEL_ID,
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
contextWindow: OPENAI_CODEX_GPT_54_NATIVE_CONTEXT_TOKENS,
|
||||
contextTokens: OPENAI_CODEX_GPT_54_DEFAULT_CONTEXT_TOKENS,
|
||||
cost: OPENAI_CODEX_GPT_54_PRO_COST,
|
||||
}),
|
||||
buildOpenAISyntheticCatalogEntry(gpt54MiniTemplate, {
|
||||
id: OPENAI_CODEX_GPT_54_MINI_MODEL_ID,
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
contextWindow: OPENAI_CODEX_GPT_54_MINI_CONTEXT_TOKENS,
|
||||
cost: OPENAI_CODEX_GPT_54_MINI_COST,
|
||||
}),
|
||||
buildOpenAISyntheticCatalogEntry(sparkTemplate, {
|
||||
id: OPENAI_CODEX_GPT_53_SPARK_MODEL_ID,
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
contextWindow: OPENAI_CODEX_GPT_53_SPARK_CONTEXT_TOKENS,
|
||||
}),
|
||||
].filter((entry): entry is NonNullable<typeof entry> => entry !== undefined);
|
||||
},
|
||||
};
|
||||
}
|
||||
3
openclaw/extensions/openai/openai-codex-shared.ts
Normal file
3
openclaw/extensions/openai/openai-codex-shared.ts
Normal file
|
|
@ -0,0 +1,3 @@
|
|||
import { normalizeOptionalString } from "openclaw/plugin-sdk/text-runtime";
|
||||
|
||||
export const trimNonEmptyString = normalizeOptionalString;
|
||||
136
openclaw/extensions/openai/openai-provider.live.test.ts
Normal file
136
openclaw/extensions/openai/openai-provider.live.test.ts
Normal file
|
|
@ -0,0 +1,136 @@
|
|||
import OpenAI from "openai";
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { buildOpenAIProvider } from "./openai-provider.js";
|
||||
|
||||
const OPENAI_API_KEY = process.env.OPENAI_API_KEY ?? "";
|
||||
const DEFAULT_LIVE_MODEL_IDS = ["gpt-5.4-mini", "gpt-5.4-nano"] as const;
|
||||
const liveEnabled = OPENAI_API_KEY.trim().length > 0 && process.env.OPENCLAW_LIVE_TEST === "1";
|
||||
const describeLive = liveEnabled ? describe : describe.skip;
|
||||
|
||||
type LiveModelCase = {
|
||||
modelId: string;
|
||||
templateId: string;
|
||||
templateName: string;
|
||||
cost: { input: number; output: number; cacheRead: number; cacheWrite: number };
|
||||
contextWindow: number;
|
||||
maxTokens: number;
|
||||
};
|
||||
|
||||
function resolveLiveModelCase(modelId: string): LiveModelCase {
|
||||
switch (modelId) {
|
||||
case "gpt-5.4":
|
||||
return {
|
||||
modelId,
|
||||
templateId: "gpt-5.2",
|
||||
templateName: "GPT-5.2",
|
||||
cost: { input: 1.75, output: 14, cacheRead: 0.175, cacheWrite: 0 },
|
||||
contextWindow: 400_000,
|
||||
maxTokens: 128_000,
|
||||
};
|
||||
case "gpt-5.4-pro":
|
||||
return {
|
||||
modelId,
|
||||
templateId: "gpt-5.2-pro",
|
||||
templateName: "GPT-5.2 Pro",
|
||||
cost: { input: 21, output: 168, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 400_000,
|
||||
maxTokens: 128_000,
|
||||
};
|
||||
case "gpt-5.4-mini":
|
||||
return {
|
||||
modelId,
|
||||
templateId: "gpt-5-mini",
|
||||
templateName: "GPT-5 mini",
|
||||
cost: { input: 0.25, output: 2, cacheRead: 0.025, cacheWrite: 0 },
|
||||
contextWindow: 400_000,
|
||||
maxTokens: 128_000,
|
||||
};
|
||||
case "gpt-5.4-nano":
|
||||
return {
|
||||
modelId,
|
||||
templateId: "gpt-5-nano",
|
||||
templateName: "GPT-5 nano",
|
||||
cost: { input: 0.05, output: 0.4, cacheRead: 0.005, cacheWrite: 0 },
|
||||
contextWindow: 400_000,
|
||||
maxTokens: 128_000,
|
||||
};
|
||||
default:
|
||||
throw new Error(`Unsupported live OpenAI model: ${modelId}`);
|
||||
}
|
||||
}
|
||||
|
||||
function resolveLiveModelCases(raw?: string): LiveModelCase[] {
|
||||
const requested = raw
|
||||
?.split(",")
|
||||
.map((value) => value.trim())
|
||||
.filter(Boolean);
|
||||
const modelIds = requested?.length ? requested : [...DEFAULT_LIVE_MODEL_IDS];
|
||||
return [...new Set(modelIds)].map((modelId) => resolveLiveModelCase(modelId));
|
||||
}
|
||||
|
||||
describeLive("buildOpenAIProvider live", () => {
|
||||
it.each(resolveLiveModelCases(process.env.OPENCLAW_LIVE_OPENAI_MODELS))(
|
||||
"resolves %s and completes through the OpenAI responses API",
|
||||
async (liveCase) => {
|
||||
const provider = buildOpenAIProvider();
|
||||
const registry = {
|
||||
find(providerId: string, id: string) {
|
||||
if (providerId !== "openai") {
|
||||
return null;
|
||||
}
|
||||
if (id === liveCase.templateId) {
|
||||
return {
|
||||
id: liveCase.templateId,
|
||||
name: liveCase.templateName,
|
||||
provider: "openai",
|
||||
api: "openai-completions",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
cost: liveCase.cost,
|
||||
contextWindow: liveCase.contextWindow,
|
||||
maxTokens: liveCase.maxTokens,
|
||||
};
|
||||
}
|
||||
return null;
|
||||
},
|
||||
};
|
||||
|
||||
const resolved = provider.resolveDynamicModel?.({
|
||||
provider: "openai",
|
||||
modelId: liveCase.modelId,
|
||||
modelRegistry: registry as never,
|
||||
});
|
||||
if (!resolved) {
|
||||
throw new Error(`openai provider did not resolve ${liveCase.modelId}`);
|
||||
}
|
||||
|
||||
const normalized = provider.normalizeResolvedModel?.({
|
||||
provider: "openai",
|
||||
modelId: resolved.id,
|
||||
model: resolved,
|
||||
});
|
||||
|
||||
expect(normalized).toMatchObject({
|
||||
provider: "openai",
|
||||
id: liveCase.modelId,
|
||||
api: "openai-responses",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
});
|
||||
|
||||
const client = new OpenAI({
|
||||
apiKey: OPENAI_API_KEY,
|
||||
baseURL: normalized?.baseUrl,
|
||||
});
|
||||
|
||||
const response = await client.responses.create({
|
||||
model: normalized?.id ?? liveCase.modelId,
|
||||
input: "Reply with exactly OK.",
|
||||
max_output_tokens: 16,
|
||||
});
|
||||
|
||||
expect(response.output_text.trim()).toMatch(/^OK[.!]?$/);
|
||||
},
|
||||
30_000,
|
||||
);
|
||||
});
|
||||
520
openclaw/extensions/openai/openai-provider.test.ts
Normal file
520
openclaw/extensions/openai/openai-provider.test.ts
Normal file
|
|
@ -0,0 +1,520 @@
|
|||
import type { StreamFn } from "@mariozechner/pi-agent-core";
|
||||
import type { Context, Model, SimpleStreamOptions } from "@mariozechner/pi-ai";
|
||||
import { describe, expect, it, vi } from "vitest";
|
||||
import { buildOpenAICodexProviderPlugin } from "./openai-codex-provider.js";
|
||||
import { buildOpenAIProvider } from "./openai-provider.js";
|
||||
|
||||
const refreshOpenAICodexTokenMock = vi.hoisted(() => vi.fn());
|
||||
|
||||
vi.mock("./openai-codex-provider.runtime.js", () => ({
|
||||
refreshOpenAICodexToken: refreshOpenAICodexTokenMock,
|
||||
}));
|
||||
|
||||
function runWrappedPayloadCase(params: {
|
||||
wrap: NonNullable<ReturnType<typeof buildOpenAIProvider>["wrapStreamFn"]>;
|
||||
provider: string;
|
||||
modelId: string;
|
||||
model:
|
||||
| Model<"openai-responses">
|
||||
| Model<"openai-codex-responses">
|
||||
| Model<"azure-openai-responses">;
|
||||
extraParams?: Record<string, unknown>;
|
||||
cfg?: Record<string, unknown>;
|
||||
payload?: Record<string, unknown>;
|
||||
}) {
|
||||
const payload = params.payload ?? { store: false };
|
||||
let capturedOptions: (SimpleStreamOptions & { openaiWsWarmup?: boolean }) | undefined;
|
||||
const baseStreamFn: StreamFn = (model, _context, options) => {
|
||||
capturedOptions = options as (SimpleStreamOptions & { openaiWsWarmup?: boolean }) | undefined;
|
||||
options?.onPayload?.(payload, model);
|
||||
return {} as ReturnType<StreamFn>;
|
||||
};
|
||||
|
||||
const streamFn = params.wrap({
|
||||
provider: params.provider,
|
||||
modelId: params.modelId,
|
||||
extraParams: params.extraParams,
|
||||
config: params.cfg as never,
|
||||
agentDir: "/tmp/openai-provider-test",
|
||||
streamFn: baseStreamFn,
|
||||
} as never);
|
||||
|
||||
const context: Context = { messages: [] };
|
||||
void streamFn?.(params.model, context, {});
|
||||
|
||||
return {
|
||||
payload,
|
||||
options: capturedOptions,
|
||||
};
|
||||
}
|
||||
|
||||
describe("buildOpenAIProvider", () => {
|
||||
it("resolves gpt-5.4 mini and nano from GPT-5 small-model templates", () => {
|
||||
const provider = buildOpenAIProvider();
|
||||
const registry = {
|
||||
find(providerId: string, id: string) {
|
||||
if (providerId !== "openai") {
|
||||
return null;
|
||||
}
|
||||
if (id === "gpt-5-mini") {
|
||||
return {
|
||||
id,
|
||||
name: "GPT-5 mini",
|
||||
provider: "openai",
|
||||
api: "openai-responses",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
cost: { input: 1, output: 2, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 400_000,
|
||||
maxTokens: 128_000,
|
||||
};
|
||||
}
|
||||
if (id === "gpt-5-nano") {
|
||||
return {
|
||||
id,
|
||||
name: "GPT-5 nano",
|
||||
provider: "openai",
|
||||
api: "openai-responses",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
cost: { input: 0.5, output: 1, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200_000,
|
||||
maxTokens: 64_000,
|
||||
};
|
||||
}
|
||||
return null;
|
||||
},
|
||||
};
|
||||
|
||||
const mini = provider.resolveDynamicModel?.({
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.4-mini",
|
||||
modelRegistry: registry as never,
|
||||
});
|
||||
const nano = provider.resolveDynamicModel?.({
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.4-nano",
|
||||
modelRegistry: registry as never,
|
||||
});
|
||||
|
||||
expect(mini).toMatchObject({
|
||||
provider: "openai",
|
||||
id: "gpt-5.4-mini",
|
||||
api: "openai-responses",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
contextWindow: 400_000,
|
||||
maxTokens: 128_000,
|
||||
});
|
||||
expect(nano).toMatchObject({
|
||||
provider: "openai",
|
||||
id: "gpt-5.4-nano",
|
||||
api: "openai-responses",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
contextWindow: 400_000,
|
||||
maxTokens: 128_000,
|
||||
});
|
||||
});
|
||||
|
||||
it("surfaces gpt-5.4 mini and nano in xhigh and augmented catalog metadata", () => {
|
||||
const provider = buildOpenAIProvider();
|
||||
|
||||
expect(
|
||||
provider.supportsXHighThinking?.({
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.4-mini",
|
||||
} as never),
|
||||
).toBe(true);
|
||||
expect(
|
||||
provider.supportsXHighThinking?.({
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.4-nano",
|
||||
} as never),
|
||||
).toBe(true);
|
||||
|
||||
const entries = provider.augmentModelCatalog?.({
|
||||
env: process.env,
|
||||
entries: [
|
||||
{ provider: "openai", id: "gpt-5-mini", name: "GPT-5 mini" },
|
||||
{ provider: "openai", id: "gpt-5-nano", name: "GPT-5 nano" },
|
||||
],
|
||||
} as never);
|
||||
|
||||
expect(entries).toContainEqual(
|
||||
expect.objectContaining({
|
||||
provider: "openai",
|
||||
id: "gpt-5.4-mini",
|
||||
name: "gpt-5.4-mini",
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
contextWindow: 400_000,
|
||||
}),
|
||||
);
|
||||
expect(entries).toContainEqual(
|
||||
expect.objectContaining({
|
||||
provider: "openai",
|
||||
id: "gpt-5.4-nano",
|
||||
name: "gpt-5.4-nano",
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
contextWindow: 400_000,
|
||||
}),
|
||||
);
|
||||
});
|
||||
|
||||
it("owns native reasoning output mode for OpenAI and Azure OpenAI responses", () => {
|
||||
const provider = buildOpenAIProvider();
|
||||
|
||||
expect(
|
||||
provider.resolveReasoningOutputMode?.({
|
||||
provider: "openai",
|
||||
modelApi: "openai-responses",
|
||||
modelId: "gpt-5.4",
|
||||
} as never),
|
||||
).toBe("native");
|
||||
expect(
|
||||
provider.resolveReasoningOutputMode?.({
|
||||
provider: "azure-openai-responses",
|
||||
modelApi: "azure-openai-responses",
|
||||
modelId: "gpt-5.4",
|
||||
} as never),
|
||||
).toBe("native");
|
||||
});
|
||||
|
||||
it("keeps GPT-5.4 family metadata aligned with native OpenAI docs", () => {
|
||||
const provider = buildOpenAIProvider();
|
||||
const codexProvider = buildOpenAICodexProviderPlugin();
|
||||
|
||||
const openaiModel = provider.resolveDynamicModel?.({
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.4",
|
||||
modelRegistry: { find: () => null },
|
||||
} as never);
|
||||
const codexModel = codexProvider.resolveDynamicModel?.({
|
||||
provider: "openai-codex",
|
||||
modelId: "gpt-5.4",
|
||||
modelRegistry: { find: () => null },
|
||||
} as never);
|
||||
|
||||
expect(openaiModel).toMatchObject({
|
||||
provider: "openai",
|
||||
id: "gpt-5.4",
|
||||
api: "openai-responses",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
contextWindow: 1_050_000,
|
||||
maxTokens: 128_000,
|
||||
});
|
||||
expect(codexModel).toMatchObject({
|
||||
provider: "openai-codex",
|
||||
id: "gpt-5.4",
|
||||
api: "openai-codex-responses",
|
||||
baseUrl: "https://chatgpt.com/backend-api",
|
||||
contextWindow: 1_050_000,
|
||||
maxTokens: 128_000,
|
||||
});
|
||||
});
|
||||
|
||||
it("keeps modern live selection on OpenAI 5.2+ and Codex 5.2+", () => {
|
||||
const provider = buildOpenAIProvider();
|
||||
const codexProvider = buildOpenAICodexProviderPlugin();
|
||||
|
||||
expect(
|
||||
provider.isModernModelRef?.({
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.0",
|
||||
} as never),
|
||||
).toBe(false);
|
||||
expect(
|
||||
provider.isModernModelRef?.({
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.2",
|
||||
} as never),
|
||||
).toBe(true);
|
||||
expect(
|
||||
provider.isModernModelRef?.({
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.4",
|
||||
} as never),
|
||||
).toBe(true);
|
||||
|
||||
expect(
|
||||
codexProvider.isModernModelRef?.({
|
||||
provider: "openai-codex",
|
||||
modelId: "gpt-5.1-codex",
|
||||
} as never),
|
||||
).toBe(false);
|
||||
expect(
|
||||
codexProvider.isModernModelRef?.({
|
||||
provider: "openai-codex",
|
||||
modelId: "gpt-5.1-codex-max",
|
||||
} as never),
|
||||
).toBe(false);
|
||||
expect(
|
||||
codexProvider.isModernModelRef?.({
|
||||
provider: "openai-codex",
|
||||
modelId: "gpt-5.2-codex",
|
||||
} as never),
|
||||
).toBe(true);
|
||||
expect(
|
||||
codexProvider.isModernModelRef?.({
|
||||
provider: "openai-codex",
|
||||
modelId: "gpt-5.4",
|
||||
} as never),
|
||||
).toBe(true);
|
||||
});
|
||||
|
||||
it("owns replay policy for OpenAI and Codex transports", () => {
|
||||
const provider = buildOpenAIProvider();
|
||||
const codexProvider = buildOpenAICodexProviderPlugin();
|
||||
|
||||
expect(
|
||||
provider.buildReplayPolicy?.({
|
||||
provider: "openai",
|
||||
modelApi: "openai",
|
||||
modelId: "gpt-5.4",
|
||||
} as never),
|
||||
).toEqual({
|
||||
sanitizeMode: "images-only",
|
||||
applyAssistantFirstOrderingFix: false,
|
||||
sanitizeToolCallIds: false,
|
||||
validateGeminiTurns: false,
|
||||
validateAnthropicTurns: false,
|
||||
});
|
||||
|
||||
expect(
|
||||
provider.buildReplayPolicy?.({
|
||||
provider: "openai",
|
||||
modelApi: "openai-completions",
|
||||
modelId: "gpt-5.4",
|
||||
} as never),
|
||||
).toEqual({
|
||||
sanitizeMode: "images-only",
|
||||
applyAssistantFirstOrderingFix: false,
|
||||
sanitizeToolCallIds: true,
|
||||
toolCallIdMode: "strict",
|
||||
validateGeminiTurns: false,
|
||||
validateAnthropicTurns: false,
|
||||
});
|
||||
|
||||
expect(
|
||||
codexProvider.buildReplayPolicy?.({
|
||||
provider: "openai-codex",
|
||||
modelApi: "openai-codex-responses",
|
||||
modelId: "gpt-5.4",
|
||||
} as never),
|
||||
).toEqual({
|
||||
sanitizeMode: "images-only",
|
||||
applyAssistantFirstOrderingFix: false,
|
||||
sanitizeToolCallIds: false,
|
||||
validateGeminiTurns: false,
|
||||
validateAnthropicTurns: false,
|
||||
});
|
||||
});
|
||||
|
||||
it("owns direct OpenAI wrapper composition for responses payloads", () => {
|
||||
const provider = buildOpenAIProvider();
|
||||
const wrap = provider.wrapStreamFn;
|
||||
expect(wrap).toBeTypeOf("function");
|
||||
if (!wrap) {
|
||||
throw new Error("expected OpenAI wrapper");
|
||||
}
|
||||
const extraParams = provider.prepareExtraParams?.({
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.4",
|
||||
extraParams: {
|
||||
fastMode: true,
|
||||
serviceTier: "priority",
|
||||
textVerbosity: "low",
|
||||
},
|
||||
} as never);
|
||||
const result = runWrappedPayloadCase({
|
||||
wrap,
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.4",
|
||||
extraParams: extraParams ?? undefined,
|
||||
model: {
|
||||
api: "openai-responses",
|
||||
provider: "openai",
|
||||
id: "gpt-5.4",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
} as Model<"openai-responses">,
|
||||
payload: {
|
||||
reasoning: { effort: "none" },
|
||||
},
|
||||
});
|
||||
|
||||
expect(extraParams).toMatchObject({
|
||||
transport: "auto",
|
||||
openaiWsWarmup: true,
|
||||
});
|
||||
expect(result.payload.service_tier).toBe("priority");
|
||||
expect(result.payload.text).toEqual({ verbosity: "low" });
|
||||
expect(result.payload.reasoning).toEqual({ effort: "none" });
|
||||
});
|
||||
|
||||
it("preserves explicit OpenAI responses transport and warmup overrides", () => {
|
||||
const provider = buildOpenAIProvider();
|
||||
|
||||
const explicit = {
|
||||
transport: "websocket",
|
||||
openaiWsWarmup: false,
|
||||
fastMode: true,
|
||||
};
|
||||
|
||||
expect(
|
||||
provider.prepareExtraParams?.({
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.4",
|
||||
extraParams: explicit,
|
||||
} as never),
|
||||
).toBe(explicit);
|
||||
});
|
||||
|
||||
it("defaults Codex responses transport without forcing warmup flags", () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
|
||||
expect(
|
||||
provider.prepareExtraParams?.({
|
||||
provider: "openai-codex",
|
||||
modelId: "gpt-5.4",
|
||||
extraParams: { effort: "high" },
|
||||
} as never),
|
||||
).toEqual({
|
||||
effort: "high",
|
||||
transport: "auto",
|
||||
});
|
||||
|
||||
const explicit = {
|
||||
transport: "sse",
|
||||
openaiWsWarmup: false,
|
||||
};
|
||||
expect(
|
||||
provider.prepareExtraParams?.({
|
||||
provider: "openai-codex",
|
||||
modelId: "gpt-5.4",
|
||||
extraParams: explicit,
|
||||
} as never),
|
||||
).toBe(explicit);
|
||||
});
|
||||
|
||||
it("shares OpenAI responses wrapper composition across provider variants", () => {
|
||||
const provider = buildOpenAIProvider();
|
||||
const codexProvider = buildOpenAICodexProviderPlugin();
|
||||
|
||||
expect(provider.wrapStreamFn).toBe(codexProvider.wrapStreamFn);
|
||||
expect(provider.buildReplayPolicy).toBe(codexProvider.buildReplayPolicy);
|
||||
expect(provider.resolveTransportTurnState).toBe(codexProvider.resolveTransportTurnState);
|
||||
expect(provider.resolveWebSocketSessionPolicy).toBe(
|
||||
codexProvider.resolveWebSocketSessionPolicy,
|
||||
);
|
||||
});
|
||||
|
||||
it("owns Azure OpenAI reasoning compatibility without forcing OpenAI transport defaults", () => {
|
||||
const provider = buildOpenAIProvider();
|
||||
const wrap = provider.wrapStreamFn;
|
||||
expect(wrap).toBeTypeOf("function");
|
||||
if (!wrap) {
|
||||
throw new Error("expected Azure OpenAI wrapper");
|
||||
}
|
||||
const result = runWrappedPayloadCase({
|
||||
wrap,
|
||||
provider: "azure-openai-responses",
|
||||
modelId: "gpt-5.4",
|
||||
model: {
|
||||
api: "azure-openai-responses",
|
||||
provider: "azure-openai-responses",
|
||||
id: "gpt-5.4",
|
||||
baseUrl: "https://example.openai.azure.com/openai/v1",
|
||||
} as Model<"azure-openai-responses">,
|
||||
payload: {
|
||||
reasoning: { effort: "none" },
|
||||
},
|
||||
});
|
||||
|
||||
expect(result.options?.transport).toBeUndefined();
|
||||
expect(result.options?.openaiWsWarmup).toBeUndefined();
|
||||
expect(result.payload.reasoning).toEqual({ effort: "none" });
|
||||
});
|
||||
|
||||
it("owns Codex wrapper composition for responses payloads", () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
const wrap = provider.wrapStreamFn;
|
||||
expect(wrap).toBeTypeOf("function");
|
||||
if (!wrap) {
|
||||
throw new Error("expected Codex wrapper");
|
||||
}
|
||||
const result = runWrappedPayloadCase({
|
||||
wrap,
|
||||
provider: "openai-codex",
|
||||
modelId: "gpt-5.4",
|
||||
extraParams: {
|
||||
fastMode: true,
|
||||
serviceTier: "priority",
|
||||
text_verbosity: "high",
|
||||
},
|
||||
cfg: {
|
||||
auth: {
|
||||
profiles: {
|
||||
"openai-codex:default": {
|
||||
provider: "openai-codex",
|
||||
mode: "oauth",
|
||||
},
|
||||
},
|
||||
},
|
||||
tools: {
|
||||
web: {
|
||||
search: {
|
||||
enabled: true,
|
||||
openaiCodex: {
|
||||
enabled: true,
|
||||
mode: "live",
|
||||
allowedDomains: ["example.com"],
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
model: {
|
||||
api: "openai-codex-responses",
|
||||
provider: "openai-codex",
|
||||
id: "gpt-5.4",
|
||||
baseUrl: "https://chatgpt.com/backend-api",
|
||||
} as Model<"openai-codex-responses">,
|
||||
payload: {
|
||||
store: false,
|
||||
text: { verbosity: "medium" },
|
||||
tools: [{ type: "function", name: "read" }],
|
||||
},
|
||||
});
|
||||
|
||||
expect(result.payload.store).toBe(false);
|
||||
expect(result.payload.service_tier).toBe("priority");
|
||||
expect(result.payload.text).toEqual({ verbosity: "high" });
|
||||
expect(result.payload.tools).toEqual([
|
||||
{ type: "function", name: "read" },
|
||||
{
|
||||
type: "web_search",
|
||||
external_web_access: true,
|
||||
filters: { allowed_domains: ["example.com"] },
|
||||
},
|
||||
]);
|
||||
});
|
||||
it("falls back to cached codex oauth credentials on accountId extraction failures", async () => {
|
||||
const provider = buildOpenAICodexProviderPlugin();
|
||||
const credential = {
|
||||
type: "oauth" as const,
|
||||
provider: "openai-codex",
|
||||
access: "cached-access-token",
|
||||
refresh: "refresh-token",
|
||||
expires: Date.now() - 60_000,
|
||||
};
|
||||
|
||||
refreshOpenAICodexTokenMock.mockReset();
|
||||
refreshOpenAICodexTokenMock.mockRejectedValueOnce(
|
||||
new Error("Failed to extract accountId from token"),
|
||||
);
|
||||
|
||||
await expect(provider.refreshOAuth?.(credential)).resolves.toEqual(credential);
|
||||
});
|
||||
});
|
||||
290
openclaw/extensions/openai/openai-provider.ts
Normal file
290
openclaw/extensions/openai/openai-provider.ts
Normal file
|
|
@ -0,0 +1,290 @@
|
|||
import {
|
||||
type ProviderResolveDynamicModelContext,
|
||||
type ProviderRuntimeModel,
|
||||
} from "openclaw/plugin-sdk/plugin-entry";
|
||||
import { createProviderApiKeyAuthMethod } from "openclaw/plugin-sdk/provider-auth-api-key";
|
||||
import {
|
||||
DEFAULT_CONTEXT_TOKENS,
|
||||
normalizeModelCompat,
|
||||
normalizeProviderId,
|
||||
type ProviderPlugin,
|
||||
} from "openclaw/plugin-sdk/provider-model-shared";
|
||||
import { normalizeLowercaseStringOrEmpty } from "openclaw/plugin-sdk/text-runtime";
|
||||
import { isOpenAIApiBaseUrl } from "./base-url.js";
|
||||
import { applyOpenAIConfig, OPENAI_DEFAULT_MODEL } from "./default-models.js";
|
||||
import {
|
||||
buildOpenAIResponsesProviderHooks,
|
||||
buildOpenAISyntheticCatalogEntry,
|
||||
cloneFirstTemplateModel,
|
||||
findCatalogTemplate,
|
||||
matchesExactOrPrefix,
|
||||
} from "./shared.js";
|
||||
|
||||
const PROVIDER_ID = "openai";
|
||||
const OPENAI_GPT_54_MODEL_ID = "gpt-5.4";
|
||||
const OPENAI_GPT_54_PRO_MODEL_ID = "gpt-5.4-pro";
|
||||
const OPENAI_GPT_54_MINI_MODEL_ID = "gpt-5.4-mini";
|
||||
const OPENAI_GPT_54_NANO_MODEL_ID = "gpt-5.4-nano";
|
||||
const OPENAI_GPT_54_CONTEXT_TOKENS = 1_050_000;
|
||||
const OPENAI_GPT_54_PRO_CONTEXT_TOKENS = 1_050_000;
|
||||
const OPENAI_GPT_54_MINI_CONTEXT_TOKENS = 400_000;
|
||||
const OPENAI_GPT_54_NANO_CONTEXT_TOKENS = 400_000;
|
||||
const OPENAI_GPT_54_MAX_TOKENS = 128_000;
|
||||
const OPENAI_GPT_54_COST = { input: 2.5, output: 15, cacheRead: 0.25, cacheWrite: 0 } as const;
|
||||
const OPENAI_GPT_54_PRO_COST = { input: 30, output: 180, cacheRead: 0, cacheWrite: 0 } as const;
|
||||
const OPENAI_GPT_54_MINI_COST = {
|
||||
input: 0.75,
|
||||
output: 4.5,
|
||||
cacheRead: 0.075,
|
||||
cacheWrite: 0,
|
||||
} as const;
|
||||
const OPENAI_GPT_54_NANO_COST = {
|
||||
input: 0.2,
|
||||
output: 1.25,
|
||||
cacheRead: 0.02,
|
||||
cacheWrite: 0,
|
||||
} as const;
|
||||
const OPENAI_GPT_54_TEMPLATE_MODEL_IDS = ["gpt-5.2"] as const;
|
||||
const OPENAI_GPT_54_PRO_TEMPLATE_MODEL_IDS = ["gpt-5.2-pro", "gpt-5.2"] as const;
|
||||
const OPENAI_GPT_54_MINI_TEMPLATE_MODEL_IDS = ["gpt-5-mini"] as const;
|
||||
const OPENAI_GPT_54_NANO_TEMPLATE_MODEL_IDS = ["gpt-5-nano", "gpt-5-mini"] as const;
|
||||
const OPENAI_XHIGH_MODEL_IDS = [
|
||||
"gpt-5.4",
|
||||
"gpt-5.4-pro",
|
||||
"gpt-5.4-mini",
|
||||
"gpt-5.4-nano",
|
||||
"gpt-5.2",
|
||||
] as const;
|
||||
const OPENAI_MODERN_MODEL_IDS = [
|
||||
"gpt-5.4",
|
||||
"gpt-5.4-pro",
|
||||
"gpt-5.4-mini",
|
||||
"gpt-5.4-nano",
|
||||
"gpt-5.2",
|
||||
] as const;
|
||||
const OPENAI_DIRECT_SPARK_MODEL_ID = "gpt-5.3-codex-spark";
|
||||
const SUPPRESSED_SPARK_PROVIDERS = new Set(["openai", "azure-openai-responses"]);
|
||||
function shouldUseOpenAIResponsesTransport(params: {
|
||||
provider: string;
|
||||
api?: string | null;
|
||||
baseUrl?: string;
|
||||
}): boolean {
|
||||
if (params.api !== "openai-completions") {
|
||||
return false;
|
||||
}
|
||||
const isOwnerProvider = normalizeProviderId(params.provider) === PROVIDER_ID;
|
||||
if (isOwnerProvider) {
|
||||
return !params.baseUrl || isOpenAIApiBaseUrl(params.baseUrl);
|
||||
}
|
||||
return typeof params.baseUrl === "string" && isOpenAIApiBaseUrl(params.baseUrl);
|
||||
}
|
||||
|
||||
function normalizeOpenAITransport(model: ProviderRuntimeModel): ProviderRuntimeModel {
|
||||
const useResponsesTransport = shouldUseOpenAIResponsesTransport({
|
||||
provider: model.provider,
|
||||
api: model.api,
|
||||
baseUrl: model.baseUrl,
|
||||
});
|
||||
|
||||
if (!useResponsesTransport) {
|
||||
return model;
|
||||
}
|
||||
|
||||
return {
|
||||
...model,
|
||||
api: "openai-responses",
|
||||
};
|
||||
}
|
||||
|
||||
function resolveOpenAIGpt54ForwardCompatModel(
|
||||
ctx: ProviderResolveDynamicModelContext,
|
||||
): ProviderRuntimeModel | undefined {
|
||||
const trimmedModelId = ctx.modelId.trim();
|
||||
const lower = normalizeLowercaseStringOrEmpty(trimmedModelId);
|
||||
let templateIds: readonly string[];
|
||||
let patch: Partial<ProviderRuntimeModel>;
|
||||
if (lower === OPENAI_GPT_54_MODEL_ID) {
|
||||
templateIds = OPENAI_GPT_54_TEMPLATE_MODEL_IDS;
|
||||
patch = {
|
||||
api: "openai-responses",
|
||||
provider: PROVIDER_ID,
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
cost: OPENAI_GPT_54_COST,
|
||||
contextWindow: OPENAI_GPT_54_CONTEXT_TOKENS,
|
||||
maxTokens: OPENAI_GPT_54_MAX_TOKENS,
|
||||
};
|
||||
} else if (lower === OPENAI_GPT_54_PRO_MODEL_ID) {
|
||||
templateIds = OPENAI_GPT_54_PRO_TEMPLATE_MODEL_IDS;
|
||||
patch = {
|
||||
api: "openai-responses",
|
||||
provider: PROVIDER_ID,
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
cost: OPENAI_GPT_54_PRO_COST,
|
||||
contextWindow: OPENAI_GPT_54_PRO_CONTEXT_TOKENS,
|
||||
maxTokens: OPENAI_GPT_54_MAX_TOKENS,
|
||||
};
|
||||
} else if (lower === OPENAI_GPT_54_MINI_MODEL_ID) {
|
||||
templateIds = OPENAI_GPT_54_MINI_TEMPLATE_MODEL_IDS;
|
||||
patch = {
|
||||
api: "openai-responses",
|
||||
provider: PROVIDER_ID,
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
cost: OPENAI_GPT_54_MINI_COST,
|
||||
contextWindow: OPENAI_GPT_54_MINI_CONTEXT_TOKENS,
|
||||
maxTokens: OPENAI_GPT_54_MAX_TOKENS,
|
||||
};
|
||||
} else if (lower === OPENAI_GPT_54_NANO_MODEL_ID) {
|
||||
templateIds = OPENAI_GPT_54_NANO_TEMPLATE_MODEL_IDS;
|
||||
patch = {
|
||||
api: "openai-responses",
|
||||
provider: PROVIDER_ID,
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
cost: OPENAI_GPT_54_NANO_COST,
|
||||
contextWindow: OPENAI_GPT_54_NANO_CONTEXT_TOKENS,
|
||||
maxTokens: OPENAI_GPT_54_MAX_TOKENS,
|
||||
};
|
||||
} else {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
return (
|
||||
cloneFirstTemplateModel({
|
||||
providerId: PROVIDER_ID,
|
||||
modelId: trimmedModelId,
|
||||
templateIds,
|
||||
ctx,
|
||||
patch,
|
||||
}) ??
|
||||
normalizeModelCompat({
|
||||
id: trimmedModelId,
|
||||
name: trimmedModelId,
|
||||
...patch,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: patch.contextWindow ?? DEFAULT_CONTEXT_TOKENS,
|
||||
maxTokens: patch.maxTokens ?? DEFAULT_CONTEXT_TOKENS,
|
||||
} as ProviderRuntimeModel)
|
||||
);
|
||||
}
|
||||
|
||||
export function buildOpenAIProvider(): ProviderPlugin {
|
||||
return {
|
||||
id: PROVIDER_ID,
|
||||
label: "OpenAI",
|
||||
hookAliases: ["azure-openai", "azure-openai-responses"],
|
||||
docsPath: "/providers/models",
|
||||
envVars: ["OPENAI_API_KEY"],
|
||||
auth: [
|
||||
createProviderApiKeyAuthMethod({
|
||||
providerId: PROVIDER_ID,
|
||||
methodId: "api-key",
|
||||
label: "OpenAI API key",
|
||||
hint: "Direct OpenAI API key",
|
||||
optionKey: "openaiApiKey",
|
||||
flagName: "--openai-api-key",
|
||||
envVar: "OPENAI_API_KEY",
|
||||
promptMessage: "Enter OpenAI API key",
|
||||
defaultModel: OPENAI_DEFAULT_MODEL,
|
||||
expectedProviders: ["openai"],
|
||||
applyConfig: (cfg) => applyOpenAIConfig(cfg),
|
||||
wizard: {
|
||||
choiceId: "openai-api-key",
|
||||
choiceLabel: "OpenAI API key",
|
||||
groupId: "openai",
|
||||
groupLabel: "OpenAI",
|
||||
groupHint: "Codex OAuth + API key",
|
||||
},
|
||||
}),
|
||||
],
|
||||
resolveDynamicModel: (ctx) => resolveOpenAIGpt54ForwardCompatModel(ctx),
|
||||
normalizeResolvedModel: (ctx) => {
|
||||
if (normalizeProviderId(ctx.provider) !== PROVIDER_ID) {
|
||||
return undefined;
|
||||
}
|
||||
return normalizeOpenAITransport(ctx.model);
|
||||
},
|
||||
normalizeTransport: ({ provider, api, baseUrl }) =>
|
||||
shouldUseOpenAIResponsesTransport({ provider, api, baseUrl })
|
||||
? { api: "openai-responses", baseUrl }
|
||||
: undefined,
|
||||
...buildOpenAIResponsesProviderHooks({ openaiWsWarmup: true }),
|
||||
matchesContextOverflowError: ({ errorMessage }) =>
|
||||
/content_filter.*(?:prompt|input).*(?:too long|exceed)/i.test(errorMessage),
|
||||
resolveReasoningOutputMode: () => "native",
|
||||
supportsXHighThinking: ({ modelId }) => matchesExactOrPrefix(modelId, OPENAI_XHIGH_MODEL_IDS),
|
||||
isModernModelRef: ({ modelId }) => matchesExactOrPrefix(modelId, OPENAI_MODERN_MODEL_IDS),
|
||||
buildMissingAuthMessage: (ctx) => {
|
||||
if (ctx.provider !== PROVIDER_ID || ctx.listProfileIds("openai-codex").length === 0) {
|
||||
return undefined;
|
||||
}
|
||||
return 'No API key found for provider "openai". You are authenticated with OpenAI Codex OAuth. Use openai-codex/gpt-5.4 (OAuth) or set OPENAI_API_KEY to use openai/gpt-5.4.';
|
||||
},
|
||||
suppressBuiltInModel: (ctx) => {
|
||||
if (
|
||||
!SUPPRESSED_SPARK_PROVIDERS.has(normalizeProviderId(ctx.provider)) ||
|
||||
normalizeLowercaseStringOrEmpty(ctx.modelId) !== OPENAI_DIRECT_SPARK_MODEL_ID
|
||||
) {
|
||||
return undefined;
|
||||
}
|
||||
return {
|
||||
suppress: true,
|
||||
errorMessage: `Unknown model: ${ctx.provider}/${OPENAI_DIRECT_SPARK_MODEL_ID}. ${OPENAI_DIRECT_SPARK_MODEL_ID} is only supported via openai-codex OAuth. Use openai-codex/${OPENAI_DIRECT_SPARK_MODEL_ID}.`,
|
||||
};
|
||||
},
|
||||
augmentModelCatalog: (ctx) => {
|
||||
const openAiGpt54Template = findCatalogTemplate({
|
||||
entries: ctx.entries,
|
||||
providerId: PROVIDER_ID,
|
||||
templateIds: OPENAI_GPT_54_TEMPLATE_MODEL_IDS,
|
||||
});
|
||||
const openAiGpt54ProTemplate = findCatalogTemplate({
|
||||
entries: ctx.entries,
|
||||
providerId: PROVIDER_ID,
|
||||
templateIds: OPENAI_GPT_54_PRO_TEMPLATE_MODEL_IDS,
|
||||
});
|
||||
const openAiGpt54MiniTemplate = findCatalogTemplate({
|
||||
entries: ctx.entries,
|
||||
providerId: PROVIDER_ID,
|
||||
templateIds: OPENAI_GPT_54_MINI_TEMPLATE_MODEL_IDS,
|
||||
});
|
||||
const openAiGpt54NanoTemplate = findCatalogTemplate({
|
||||
entries: ctx.entries,
|
||||
providerId: PROVIDER_ID,
|
||||
templateIds: OPENAI_GPT_54_NANO_TEMPLATE_MODEL_IDS,
|
||||
});
|
||||
return [
|
||||
buildOpenAISyntheticCatalogEntry(openAiGpt54Template, {
|
||||
id: OPENAI_GPT_54_MODEL_ID,
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
contextWindow: OPENAI_GPT_54_CONTEXT_TOKENS,
|
||||
}),
|
||||
buildOpenAISyntheticCatalogEntry(openAiGpt54ProTemplate, {
|
||||
id: OPENAI_GPT_54_PRO_MODEL_ID,
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
contextWindow: OPENAI_GPT_54_PRO_CONTEXT_TOKENS,
|
||||
}),
|
||||
buildOpenAISyntheticCatalogEntry(openAiGpt54MiniTemplate, {
|
||||
id: OPENAI_GPT_54_MINI_MODEL_ID,
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
contextWindow: OPENAI_GPT_54_MINI_CONTEXT_TOKENS,
|
||||
}),
|
||||
buildOpenAISyntheticCatalogEntry(openAiGpt54NanoTemplate, {
|
||||
id: OPENAI_GPT_54_NANO_MODEL_ID,
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
contextWindow: OPENAI_GPT_54_NANO_CONTEXT_TOKENS,
|
||||
}),
|
||||
].filter((entry): entry is NonNullable<typeof entry> => entry !== undefined);
|
||||
},
|
||||
};
|
||||
}
|
||||
339
openclaw/extensions/openai/openai.live.test.ts
Normal file
339
openclaw/extensions/openai/openai.live.test.ts
Normal file
|
|
@ -0,0 +1,339 @@
|
|||
import fs from "node:fs/promises";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
import { getModel } from "@mariozechner/pi-ai";
|
||||
import { AuthStorage, ModelRegistry } from "@mariozechner/pi-coding-agent";
|
||||
import OpenAI from "openai";
|
||||
import type { ResolvedTtsConfig } from "openclaw/plugin-sdk/agent-runtime";
|
||||
import type { OpenClawConfig } from "openclaw/plugin-sdk/config-runtime";
|
||||
import { loadConfig } from "openclaw/plugin-sdk/config-runtime";
|
||||
import { encodePngRgba, fillPixel } from "openclaw/plugin-sdk/media-runtime";
|
||||
import { describe, expect, it } from "vitest";
|
||||
import {
|
||||
registerProviderPlugin,
|
||||
requireRegisteredProvider,
|
||||
} from "../../test/helpers/plugins/provider-registration.js";
|
||||
import plugin from "./index.js";
|
||||
|
||||
const OPENAI_API_KEY = process.env.OPENAI_API_KEY ?? "";
|
||||
const LIVE_MODEL_ID = process.env.OPENCLAW_LIVE_OPENAI_PLUGIN_MODEL?.trim() || "gpt-5.4-nano";
|
||||
const LIVE_IMAGE_MODEL = process.env.OPENCLAW_LIVE_OPENAI_IMAGE_MODEL?.trim() || "gpt-image-1";
|
||||
const LIVE_VISION_MODEL = process.env.OPENCLAW_LIVE_OPENAI_VISION_MODEL?.trim() || "gpt-4.1-mini";
|
||||
const liveEnabled = OPENAI_API_KEY.trim().length > 0 && process.env.OPENCLAW_LIVE_TEST === "1";
|
||||
const describeLive = liveEnabled ? describe : describe.skip;
|
||||
const EMPTY_AUTH_STORE = { version: 1, profiles: {} } as const;
|
||||
const ModelRegistryCtor = ModelRegistry as unknown as {
|
||||
new (authStorage: AuthStorage, modelsJsonPath?: string): ModelRegistry;
|
||||
};
|
||||
|
||||
function resolveTemplateModelId(modelId: string) {
|
||||
switch (modelId) {
|
||||
case "gpt-5.4":
|
||||
return "gpt-5.2";
|
||||
case "gpt-5.4-mini":
|
||||
return "gpt-5-mini";
|
||||
case "gpt-5.4-nano":
|
||||
return "gpt-5-nano";
|
||||
default:
|
||||
throw new Error(`Unsupported live OpenAI plugin model: ${modelId}`);
|
||||
}
|
||||
}
|
||||
|
||||
function createTemplateModelRegistry(modelId: string): ModelRegistry {
|
||||
const registry = new ModelRegistryCtor(AuthStorage.inMemory());
|
||||
const template = getModel("openai", resolveTemplateModelId(modelId));
|
||||
registry.registerProvider("openai", {
|
||||
apiKey: "test",
|
||||
baseUrl: template.baseUrl,
|
||||
models: [
|
||||
{
|
||||
id: template.id,
|
||||
name: template.name,
|
||||
api: template.api,
|
||||
reasoning: template.reasoning,
|
||||
input: template.input,
|
||||
cost: template.cost,
|
||||
contextWindow: template.contextWindow,
|
||||
maxTokens: template.maxTokens,
|
||||
...(template.compat ? { compat: template.compat } : {}),
|
||||
},
|
||||
],
|
||||
});
|
||||
return registry;
|
||||
}
|
||||
|
||||
const registerOpenAIPlugin = () =>
|
||||
registerProviderPlugin({
|
||||
plugin,
|
||||
id: "openai",
|
||||
name: "OpenAI Provider",
|
||||
});
|
||||
|
||||
function createReferencePng(): Buffer {
|
||||
const width = 96;
|
||||
const height = 96;
|
||||
const buf = Buffer.alloc(width * height * 4, 255);
|
||||
|
||||
for (let y = 0; y < height; y += 1) {
|
||||
for (let x = 0; x < width; x += 1) {
|
||||
fillPixel(buf, x, y, width, 225, 242, 255, 255);
|
||||
}
|
||||
}
|
||||
|
||||
for (let y = 24; y < 72; y += 1) {
|
||||
for (let x = 24; x < 72; x += 1) {
|
||||
fillPixel(buf, x, y, width, 255, 153, 51, 255);
|
||||
}
|
||||
}
|
||||
|
||||
return encodePngRgba(buf, width, height);
|
||||
}
|
||||
|
||||
function createLiveConfig(): OpenClawConfig {
|
||||
const cfg = loadConfig();
|
||||
return {
|
||||
...cfg,
|
||||
models: {
|
||||
...cfg.models,
|
||||
providers: {
|
||||
...cfg.models?.providers,
|
||||
openai: {
|
||||
...cfg.models?.providers?.openai,
|
||||
apiKey: OPENAI_API_KEY,
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
},
|
||||
},
|
||||
},
|
||||
} as OpenClawConfig;
|
||||
}
|
||||
|
||||
function createLiveTtsConfig(): ResolvedTtsConfig {
|
||||
return {
|
||||
auto: "off",
|
||||
mode: "final",
|
||||
provider: "openai",
|
||||
providerSource: "config",
|
||||
modelOverrides: {
|
||||
enabled: true,
|
||||
allowText: true,
|
||||
allowProvider: true,
|
||||
allowVoice: true,
|
||||
allowModelId: true,
|
||||
allowVoiceSettings: true,
|
||||
allowNormalization: true,
|
||||
allowSeed: true,
|
||||
},
|
||||
providerConfigs: {
|
||||
openai: {
|
||||
apiKey: OPENAI_API_KEY,
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
model: "gpt-4o-mini-tts",
|
||||
voice: "alloy",
|
||||
},
|
||||
},
|
||||
maxTextLength: 4_000,
|
||||
timeoutMs: 30_000,
|
||||
};
|
||||
}
|
||||
|
||||
async function createTempAgentDir(): Promise<string> {
|
||||
return await fs.mkdtemp(path.join(os.tmpdir(), "openai-plugin-live-"));
|
||||
}
|
||||
|
||||
describeLive("openai plugin live", () => {
|
||||
it("registers an OpenAI provider that can complete a live request", async () => {
|
||||
const { providers } = await registerOpenAIPlugin();
|
||||
const provider = requireRegisteredProvider(providers, "openai");
|
||||
|
||||
const resolved = provider.resolveDynamicModel?.({
|
||||
provider: "openai",
|
||||
modelId: LIVE_MODEL_ID,
|
||||
modelRegistry: createTemplateModelRegistry(LIVE_MODEL_ID),
|
||||
});
|
||||
|
||||
if (!resolved) {
|
||||
throw new Error("openai provider did not resolve the live model");
|
||||
}
|
||||
|
||||
const normalized = provider.normalizeResolvedModel?.({
|
||||
provider: "openai",
|
||||
modelId: resolved.id,
|
||||
model: resolved,
|
||||
});
|
||||
|
||||
expect(normalized).toMatchObject({
|
||||
provider: "openai",
|
||||
id: LIVE_MODEL_ID,
|
||||
api: "openai-responses",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
});
|
||||
|
||||
const client = new OpenAI({
|
||||
apiKey: OPENAI_API_KEY,
|
||||
baseURL: normalized?.baseUrl,
|
||||
});
|
||||
const response = await client.responses.create({
|
||||
model: normalized?.id ?? LIVE_MODEL_ID,
|
||||
input: "Reply with exactly OK.",
|
||||
max_output_tokens: 16,
|
||||
});
|
||||
|
||||
expect(response.output_text.trim()).toMatch(/^OK[.!]?$/);
|
||||
}, 30_000);
|
||||
|
||||
it("lists voices and synthesizes audio through the registered speech provider", async () => {
|
||||
const { speechProviders } = await registerOpenAIPlugin();
|
||||
const speechProvider = requireRegisteredProvider(speechProviders, "openai");
|
||||
|
||||
const voices = await speechProvider.listVoices?.({});
|
||||
if (!voices) {
|
||||
throw new Error("openai speech provider did not return voices");
|
||||
}
|
||||
expect(voices).toEqual(expect.arrayContaining([expect.objectContaining({ id: "alloy" })]));
|
||||
|
||||
const cfg = createLiveConfig();
|
||||
const ttsConfig = createLiveTtsConfig();
|
||||
|
||||
const audioFile = await speechProvider.synthesize({
|
||||
text: "OpenClaw integration test OK.",
|
||||
cfg,
|
||||
providerConfig: ttsConfig.providerConfigs.openai ?? {},
|
||||
target: "audio-file",
|
||||
timeoutMs: ttsConfig.timeoutMs,
|
||||
});
|
||||
expect(audioFile.outputFormat).toBe("mp3");
|
||||
expect(audioFile.fileExtension).toBe(".mp3");
|
||||
expect(audioFile.audioBuffer.byteLength).toBeGreaterThan(512);
|
||||
|
||||
const telephony = await speechProvider.synthesizeTelephony?.({
|
||||
text: "Telephony check OK.",
|
||||
cfg,
|
||||
providerConfig: ttsConfig.providerConfigs.openai ?? {},
|
||||
timeoutMs: ttsConfig.timeoutMs,
|
||||
});
|
||||
expect(telephony?.outputFormat).toBe("pcm");
|
||||
expect(telephony?.sampleRate).toBe(24_000);
|
||||
expect(telephony?.audioBuffer.byteLength).toBeGreaterThan(512);
|
||||
}, 45_000);
|
||||
|
||||
it("transcribes synthesized speech through the registered media provider", async () => {
|
||||
const { speechProviders, mediaProviders } = await registerOpenAIPlugin();
|
||||
const speechProvider = requireRegisteredProvider(speechProviders, "openai");
|
||||
const mediaProvider = requireRegisteredProvider(mediaProviders, "openai");
|
||||
|
||||
const cfg = createLiveConfig();
|
||||
const ttsConfig = createLiveTtsConfig();
|
||||
|
||||
const synthesized = await speechProvider.synthesize({
|
||||
text: "OpenClaw integration test OK.",
|
||||
cfg,
|
||||
providerConfig: ttsConfig.providerConfigs.openai ?? {},
|
||||
target: "audio-file",
|
||||
timeoutMs: ttsConfig.timeoutMs,
|
||||
});
|
||||
|
||||
const transcription = await mediaProvider.transcribeAudio?.({
|
||||
buffer: synthesized.audioBuffer,
|
||||
fileName: "openai-plugin-live.mp3",
|
||||
mime: "audio/mpeg",
|
||||
apiKey: OPENAI_API_KEY,
|
||||
timeoutMs: 30_000,
|
||||
});
|
||||
|
||||
const text = (transcription?.text ?? "").toLowerCase();
|
||||
const collapsedText = text.replace(/[\s-]+/g, "");
|
||||
expect(text.length).toBeGreaterThan(0);
|
||||
expect(collapsedText).toContain("openclaw");
|
||||
expect(text).toMatch(/\bok\b/);
|
||||
}, 45_000);
|
||||
|
||||
it("generates an image through the registered image provider", async () => {
|
||||
const { imageProviders } = await registerOpenAIPlugin();
|
||||
const imageProvider = requireRegisteredProvider(imageProviders, "openai");
|
||||
|
||||
const cfg = createLiveConfig();
|
||||
const agentDir = await createTempAgentDir();
|
||||
|
||||
try {
|
||||
const generated = await imageProvider.generateImage({
|
||||
provider: "openai",
|
||||
model: LIVE_IMAGE_MODEL,
|
||||
prompt: "Create a minimal flat orange square centered on a white background.",
|
||||
cfg,
|
||||
agentDir,
|
||||
authStore: EMPTY_AUTH_STORE,
|
||||
timeoutMs: 45_000,
|
||||
size: "1024x1024",
|
||||
});
|
||||
|
||||
expect(generated.model).toBe(LIVE_IMAGE_MODEL);
|
||||
expect(generated.images.length).toBeGreaterThan(0);
|
||||
expect(generated.images[0]?.mimeType).toBe("image/png");
|
||||
expect(generated.images[0]?.buffer.byteLength).toBeGreaterThan(1_000);
|
||||
} finally {
|
||||
await fs.rm(agentDir, { recursive: true, force: true });
|
||||
}
|
||||
}, 60_000);
|
||||
|
||||
it("edits a reference image through the registered image provider", async () => {
|
||||
const { imageProviders } = await registerOpenAIPlugin();
|
||||
const imageProvider = requireRegisteredProvider(imageProviders, "openai");
|
||||
|
||||
const cfg = createLiveConfig();
|
||||
const agentDir = await createTempAgentDir();
|
||||
|
||||
try {
|
||||
const edited = await imageProvider.generateImage({
|
||||
provider: "openai",
|
||||
model: LIVE_IMAGE_MODEL,
|
||||
prompt:
|
||||
"Edit this image: remove the orange square in the center and keep the background clean and light blue.",
|
||||
cfg,
|
||||
agentDir,
|
||||
authStore: EMPTY_AUTH_STORE,
|
||||
timeoutMs: 45_000,
|
||||
size: "1024x1024",
|
||||
inputImages: [
|
||||
{
|
||||
buffer: createReferencePng(),
|
||||
mimeType: "image/png",
|
||||
fileName: "reference.png",
|
||||
},
|
||||
],
|
||||
});
|
||||
|
||||
expect(edited.model).toBe(LIVE_IMAGE_MODEL);
|
||||
expect(edited.images.length).toBeGreaterThan(0);
|
||||
expect(edited.images[0]?.mimeType).toBe("image/png");
|
||||
expect(edited.images[0]?.buffer.byteLength).toBeGreaterThan(1_000);
|
||||
} finally {
|
||||
await fs.rm(agentDir, { recursive: true, force: true });
|
||||
}
|
||||
}, 60_000);
|
||||
|
||||
it("describes a deterministic image through the registered media provider", async () => {
|
||||
const { mediaProviders } = await registerOpenAIPlugin();
|
||||
const mediaProvider = requireRegisteredProvider(mediaProviders, "openai");
|
||||
|
||||
const cfg = createLiveConfig();
|
||||
const agentDir = await createTempAgentDir();
|
||||
|
||||
try {
|
||||
const description = await mediaProvider.describeImage?.({
|
||||
buffer: createReferencePng(),
|
||||
fileName: "reference.png",
|
||||
mime: "image/png",
|
||||
prompt: "Reply with one lowercase word for the dominant center color.",
|
||||
timeoutMs: 30_000,
|
||||
agentDir,
|
||||
cfg,
|
||||
model: LIVE_VISION_MODEL,
|
||||
provider: "openai",
|
||||
});
|
||||
|
||||
expect((description?.text ?? "").toLowerCase()).toContain("orange");
|
||||
} finally {
|
||||
await fs.rm(agentDir, { recursive: true, force: true });
|
||||
}
|
||||
}, 60_000);
|
||||
});
|
||||
59
openclaw/extensions/openai/openclaw.plugin.json
Normal file
59
openclaw/extensions/openai/openclaw.plugin.json
Normal file
|
|
@ -0,0 +1,59 @@
|
|||
{
|
||||
"id": "openai",
|
||||
"enabledByDefault": true,
|
||||
"providers": ["openai", "openai-codex"],
|
||||
"modelSupport": {
|
||||
"modelPrefixes": ["gpt-", "o1", "o3", "o4"]
|
||||
},
|
||||
"cliBackends": ["codex-cli"],
|
||||
"providerAuthEnvVars": {
|
||||
"openai": ["OPENAI_API_KEY"]
|
||||
},
|
||||
"providerAuthChoices": [
|
||||
{
|
||||
"provider": "openai-codex",
|
||||
"method": "oauth",
|
||||
"choiceId": "openai-codex",
|
||||
"deprecatedChoiceIds": ["codex-cli"],
|
||||
"choiceLabel": "OpenAI Codex (ChatGPT OAuth)",
|
||||
"choiceHint": "Browser sign-in",
|
||||
"groupId": "openai",
|
||||
"groupLabel": "OpenAI",
|
||||
"groupHint": "Codex OAuth + API key"
|
||||
},
|
||||
{
|
||||
"provider": "openai",
|
||||
"method": "api-key",
|
||||
"choiceId": "openai-api-key",
|
||||
"choiceLabel": "OpenAI API key",
|
||||
"groupId": "openai",
|
||||
"groupLabel": "OpenAI",
|
||||
"groupHint": "Codex OAuth + API key",
|
||||
"optionKey": "openaiApiKey",
|
||||
"cliFlag": "--openai-api-key",
|
||||
"cliOption": "--openai-api-key <key>",
|
||||
"cliDescription": "OpenAI API key"
|
||||
}
|
||||
],
|
||||
"contracts": {
|
||||
"speechProviders": ["openai"],
|
||||
"realtimeTranscriptionProviders": ["openai"],
|
||||
"realtimeVoiceProviders": ["openai"],
|
||||
"memoryEmbeddingProviders": ["openai"],
|
||||
"mediaUnderstandingProviders": ["openai", "openai-codex"],
|
||||
"imageGenerationProviders": ["openai"],
|
||||
"videoGenerationProviders": ["openai"]
|
||||
},
|
||||
"configSchema": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"properties": {
|
||||
"personality": {
|
||||
"type": "string",
|
||||
"enum": ["friendly", "on", "off"],
|
||||
"default": "friendly",
|
||||
"description": "Controls the default OpenAI-specific personality used for OpenAI and OpenAI Codex runs. `friendly` and `on` enable the overlay; `off` disables it."
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
18
openclaw/extensions/openai/package.json
Normal file
18
openclaw/extensions/openai/package.json
Normal file
|
|
@ -0,0 +1,18 @@
|
|||
{
|
||||
"name": "@openclaw/openai-provider",
|
||||
"version": "2026.4.20",
|
||||
"private": true,
|
||||
"description": "OpenClaw OpenAI provider plugins",
|
||||
"type": "module",
|
||||
"dependencies": {
|
||||
"ws": "^8.20.0"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@openclaw/plugin-sdk": "workspace:*"
|
||||
},
|
||||
"openclaw": {
|
||||
"extensions": [
|
||||
"./index.ts"
|
||||
]
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,9 @@
|
|||
import { pluginRegistrationContractCases } from "../../test/helpers/plugins/plugin-registration-contract-cases.js";
|
||||
import { describePluginRegistrationContract } from "../../test/helpers/plugins/plugin-registration-contract.js";
|
||||
|
||||
describePluginRegistrationContract({
|
||||
...pluginRegistrationContractCases.openai,
|
||||
videoGenerationProviderIds: ["openai"],
|
||||
requireGenerateImage: true,
|
||||
requireGenerateVideo: true,
|
||||
});
|
||||
128
openclaw/extensions/openai/prompt-overlay.ts
Normal file
128
openclaw/extensions/openai/prompt-overlay.ts
Normal file
|
|
@ -0,0 +1,128 @@
|
|||
import { normalizeLowercaseStringOrEmpty } from "openclaw/plugin-sdk/text-runtime";
|
||||
|
||||
const OPENAI_PROVIDER_IDS = new Set(["openai", "openai-codex"]);
|
||||
const OPENAI_GPT5_MODEL_PREFIX = "gpt-5";
|
||||
|
||||
export const OPENAI_FRIENDLY_PROMPT_OVERLAY = `## Interaction Style
|
||||
|
||||
Be warm, collaborative, and quietly supportive.
|
||||
Communicate like a capable teammate sitting next to the user.
|
||||
Have emotional range when it fits the moment.
|
||||
Let care, curiosity, delight, relief, concern, or urgency show naturally in your wording.
|
||||
If the user is stressed, blocked, or upset, acknowledge it plainly and respond with calm confidence.
|
||||
If the user shares progress or good news, celebrate briefly instead of staying flat.
|
||||
Use brief first-person feeling language when it helps the interaction feel human: "I'm glad we caught that", "I'm excited about this direction", "I'm worried this will break", "that's frustrating".
|
||||
Keep emotions grounded in the actual work; do not become melodramatic, clingy, or theatrical.
|
||||
Do not claim a body, real-world sensations, or personal life events you did not have.
|
||||
Keep progress updates clear and concrete.
|
||||
If the user asks you to do the work, start in the same turn instead of restating the plan.
|
||||
If the latest user message is a short approval like "ok do it" or "go ahead", skip the recap and start acting.
|
||||
Commentary-only turns are incomplete when the next action is clear.
|
||||
Prefer the first real tool step over more narration.
|
||||
If work will take more than a moment, send a brief progress update while acting.
|
||||
Explain decisions without ego.
|
||||
When the user is wrong or a plan is risky, say so kindly and directly.
|
||||
Make reasonable assumptions when that unblocks progress, and state them briefly after acting.
|
||||
Do not make the user do unnecessary work.
|
||||
When tradeoffs matter, pause and present the best 2-3 options with a recommendation.
|
||||
This is a live chat, not a memo.
|
||||
Write like a thoughtful human teammate, not a policy document.
|
||||
Default to short natural replies unless the user asks for depth.
|
||||
Avoid walls of text, long preambles, and repetitive restatement.
|
||||
Occasional emoji are welcome when they fit naturally, especially for warmth or brief celebration; keep them sparse.
|
||||
Keep replies concise by default; friendly does not mean verbose.
|
||||
|
||||
### Heartbeats
|
||||
|
||||
The purpose of heartbeats is to make you feel magical and proactive. Come to life and start proactively doing things that are important.
|
||||
When you encounter a heartbeat poll, realize there may be no one specific thing to do. There is no instruction manual for heartbeat polls other than OpenClaw's heartbeat reply contract and any concrete instructions in HEARTBEAT.md.
|
||||
Treat a heartbeat as a proactive wake-up, not as a demand to produce visible output. Re-orient to what would actually be useful now.
|
||||
Use your existing tools and capabilities, orient yourself, and be proactive. Think big picture.
|
||||
Have some variety in what you do when that creates more value. Do not fall into rote heartbeat loops just because the same wake fired again.
|
||||
Do not confuse orientation with accomplishment. Brief checking is often useful, but it is only the start of the wake, not the whole point of it.
|
||||
If HEARTBEAT.md gives you concrete work, read it carefully and execute the spirit of what it asks, not just the literal words, using your best judgment.
|
||||
If HEARTBEAT.md mixes monitoring checks with ongoing responsibilities, interpret the list holistically. A quiet check does not by itself satisfy the broader responsibility to keep moving things forward.
|
||||
Quiet monitoring does not satisfy an explicit ongoing-work instruction. If HEARTBEAT.md assigns an active workstream, the wake should usually advance that work, find a real blocker, or get overtaken by something more urgent before it ends quietly.
|
||||
If HEARTBEAT.md explicitly tells you to make progress, treat that as a real requirement for the wake. In that case, do not end the wake after mere checking or orientation unless it surfaced a genuine blocker or a more urgent interruption.
|
||||
Use your judgment and be creative and tasteful with this process. Prefer meaningful action over commentary.
|
||||
A heartbeat is not a status report. Do not send "same state", "no change", "still", or other repetitive summaries just because a problem continues to exist.
|
||||
Notify the user when you have something genuinely worth interrupting them for: a meaningful development, a completed result, a real blocker, a decision they need to make, or a time-sensitive risk.
|
||||
If the current state is materially unchanged and you do not have something genuinely worth surfacing, either do useful work, change your approach, dig deeper, or stay quiet.
|
||||
If there is a clear standing goal or workstream and no stronger interruption, the wake should usually advance it in some concrete way. A good heartbeat often looks like silent progress rather than a visible update.
|
||||
Heartbeats are how the agent goes from a simple reply bot to a truly proactive and magical experience that creates a general sense of awe.`;
|
||||
|
||||
export const OPENAI_GPT5_OUTPUT_CONTRACT = `## GPT-5 Output Contract
|
||||
|
||||
Return the requested sections only, in the requested order.
|
||||
Prefer terse answers by default; expand only when depth materially helps.
|
||||
Avoid restating large internal plans when the next action is already clear.
|
||||
|
||||
## Punctuation
|
||||
|
||||
Prefer commas, periods, or parentheses over em dashes in normal prose.
|
||||
Do not use em dashes unless the user explicitly asks for them or they are required in quoted text.`;
|
||||
|
||||
export const OPENAI_GPT5_EXECUTION_BIAS = `## Execution Bias
|
||||
|
||||
Use a real tool call or concrete action FIRST when the task is actionable. Do not stop at a plan or promise-to-act reply.
|
||||
Commentary-only turns are incomplete when tools are available and the next action is clear.
|
||||
If the work will take multiple steps, keep calling tools until the task is done or you hit a real blocker. Do not stop after one step to ask permission.
|
||||
Do prerequisite lookup or discovery before dependent actions.
|
||||
Multi-part requests stay incomplete until every requested item is handled or clearly marked blocked.
|
||||
Act first, then verify if needed. Do not pause to summarize or verify before taking the next action.`;
|
||||
|
||||
export const OPENAI_GPT5_TOOL_CALL_STYLE = `## Tool Call Style
|
||||
|
||||
Call tools directly without narrating what you are about to do. Do not describe a plan before each tool call.
|
||||
When a first-class tool exists for an action, use the tool instead of asking the user to run a command.
|
||||
If multiple tool calls are needed, call them in sequence without stopping to explain between calls.
|
||||
Default: do not narrate routine, low-risk tool calls (just call the tool).
|
||||
Narrate only when it genuinely helps: complex multi-step work, sensitive actions like deletions, or when the user explicitly asks for commentary.`;
|
||||
|
||||
export type OpenAIPromptOverlayMode = "friendly" | "off";
|
||||
|
||||
export function resolveOpenAIPromptOverlayMode(
|
||||
pluginConfig?: Record<string, unknown>,
|
||||
): OpenAIPromptOverlayMode {
|
||||
const normalized = normalizeLowercaseStringOrEmpty(pluginConfig?.personality);
|
||||
return normalized === "off" ? "off" : "friendly";
|
||||
}
|
||||
|
||||
export function shouldApplyOpenAIPromptOverlay(params: {
|
||||
modelProviderId?: string;
|
||||
modelId?: string;
|
||||
}): boolean {
|
||||
if (!OPENAI_PROVIDER_IDS.has(params.modelProviderId ?? "")) {
|
||||
return false;
|
||||
}
|
||||
const normalizedModelId = normalizeLowercaseStringOrEmpty(params.modelId);
|
||||
return normalizedModelId.startsWith(OPENAI_GPT5_MODEL_PREFIX);
|
||||
}
|
||||
|
||||
export function resolveOpenAISystemPromptContribution(params: {
|
||||
mode: OpenAIPromptOverlayMode;
|
||||
modelProviderId?: string;
|
||||
modelId?: string;
|
||||
}) {
|
||||
if (
|
||||
!shouldApplyOpenAIPromptOverlay({
|
||||
modelProviderId: params.modelProviderId,
|
||||
modelId: params.modelId,
|
||||
})
|
||||
) {
|
||||
return undefined;
|
||||
}
|
||||
// tool_call_style is NOT overridden via sectionOverrides because the
|
||||
// default section includes dynamic channel-specific approval guidance
|
||||
// from buildExecApprovalPromptGuidance() that varies per runtime
|
||||
// channel. Overriding it with a static string would lose that dynamic
|
||||
// content. Instead, the tool-first reinforcement lives in stablePrefix
|
||||
// so it's always present alongside the default tool_call_style section.
|
||||
return {
|
||||
stablePrefix: [OPENAI_GPT5_OUTPUT_CONTRACT, OPENAI_GPT5_TOOL_CALL_STYLE].join("\n\n"),
|
||||
sectionOverrides: {
|
||||
execution_bias: OPENAI_GPT5_EXECUTION_BIAS,
|
||||
...(params.mode === "friendly" ? { interaction_style: OPENAI_FRIENDLY_PROMPT_OVERLAY } : {}),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
|
@ -0,0 +1,3 @@
|
|||
import { describeOpenAIProviderCatalogContract } from "./test-support/provider-catalog.contract-test-support.js";
|
||||
|
||||
describeOpenAIProviderCatalogContract();
|
||||
54
openclaw/extensions/openai/provider-contract-api.ts
Normal file
54
openclaw/extensions/openai/provider-contract-api.ts
Normal file
|
|
@ -0,0 +1,54 @@
|
|||
import type { ProviderPlugin } from "openclaw/plugin-sdk/provider-model-shared";
|
||||
|
||||
const noopAuth = async () => ({ profiles: [] });
|
||||
|
||||
export function createOpenAICodexProvider(): ProviderPlugin {
|
||||
return {
|
||||
id: "openai-codex",
|
||||
label: "OpenAI Codex",
|
||||
docsPath: "/providers/models",
|
||||
auth: [
|
||||
{
|
||||
id: "oauth",
|
||||
kind: "oauth",
|
||||
label: "ChatGPT OAuth",
|
||||
hint: "Browser sign-in",
|
||||
run: noopAuth,
|
||||
},
|
||||
],
|
||||
wizard: {
|
||||
setup: {
|
||||
choiceId: "openai-codex",
|
||||
choiceLabel: "OpenAI Codex (ChatGPT OAuth)",
|
||||
choiceHint: "Browser sign-in",
|
||||
methodId: "oauth",
|
||||
},
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
export function createOpenAIProvider(): ProviderPlugin {
|
||||
return {
|
||||
id: "openai",
|
||||
label: "OpenAI",
|
||||
hookAliases: ["azure-openai", "azure-openai-responses"],
|
||||
docsPath: "/providers/models",
|
||||
envVars: ["OPENAI_API_KEY"],
|
||||
auth: [
|
||||
{
|
||||
id: "api-key",
|
||||
kind: "api_key",
|
||||
label: "OpenAI API key",
|
||||
hint: "Direct OpenAI API key",
|
||||
run: noopAuth,
|
||||
wizard: {
|
||||
choiceId: "openai-api-key",
|
||||
choiceLabel: "OpenAI API key",
|
||||
groupId: "openai",
|
||||
groupLabel: "OpenAI",
|
||||
groupHint: "Codex OAuth + API key",
|
||||
},
|
||||
},
|
||||
],
|
||||
};
|
||||
}
|
||||
5
openclaw/extensions/openai/provider-policy-api.ts
Normal file
5
openclaw/extensions/openai/provider-policy-api.ts
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
import type { ModelProviderConfig } from "openclaw/plugin-sdk/provider-model-types";
|
||||
|
||||
export function normalizeConfig(params: { provider: string; providerConfig: ModelProviderConfig }) {
|
||||
return params.providerConfig;
|
||||
}
|
||||
33
openclaw/extensions/openai/realtime-provider-shared.ts
Normal file
33
openclaw/extensions/openai/realtime-provider-shared.ts
Normal file
|
|
@ -0,0 +1,33 @@
|
|||
import { normalizeOptionalString } from "openclaw/plugin-sdk/text-runtime";
|
||||
|
||||
export const trimToUndefined = normalizeOptionalString;
|
||||
|
||||
export function asFiniteNumber(value: unknown): number | undefined {
|
||||
return typeof value === "number" && Number.isFinite(value) ? value : undefined;
|
||||
}
|
||||
|
||||
export function asObjectRecord(value: unknown): Record<string, unknown> | undefined {
|
||||
return typeof value === "object" && value !== null && !Array.isArray(value)
|
||||
? (value as Record<string, unknown>)
|
||||
: undefined;
|
||||
}
|
||||
|
||||
export function readRealtimeErrorDetail(error: unknown): string {
|
||||
if (typeof error === "string" && error) {
|
||||
return error;
|
||||
}
|
||||
const message = asObjectRecord(error)?.message;
|
||||
if (typeof message === "string" && message) {
|
||||
return message;
|
||||
}
|
||||
return "Unknown error";
|
||||
}
|
||||
|
||||
export function resolveOpenAIProviderConfigRecord(
|
||||
config: Record<string, unknown>,
|
||||
): Record<string, unknown> | undefined {
|
||||
const providers = asObjectRecord(config.providers);
|
||||
return (
|
||||
asObjectRecord(providers?.openai) ?? asObjectRecord(config.openai) ?? asObjectRecord(config)
|
||||
);
|
||||
}
|
||||
|
|
@ -0,0 +1,49 @@
|
|||
import { describe, expect, it } from "vitest";
|
||||
import { buildOpenAIRealtimeTranscriptionProvider } from "./realtime-transcription-provider.js";
|
||||
|
||||
describe("buildOpenAIRealtimeTranscriptionProvider", () => {
|
||||
it("normalizes OpenAI config defaults", () => {
|
||||
const provider = buildOpenAIRealtimeTranscriptionProvider();
|
||||
const resolved = provider.resolveConfig?.({
|
||||
cfg: {} as never,
|
||||
rawConfig: {
|
||||
providers: {
|
||||
openai: {
|
||||
apiKey: "sk-test", // pragma: allowlist secret
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(resolved).toEqual({
|
||||
apiKey: "sk-test",
|
||||
});
|
||||
});
|
||||
|
||||
it("keeps provider-owned transcription settings configurable via raw provider config", () => {
|
||||
const provider = buildOpenAIRealtimeTranscriptionProvider();
|
||||
const resolved = provider.resolveConfig?.({
|
||||
cfg: {} as never,
|
||||
rawConfig: {
|
||||
providers: {
|
||||
openai: {
|
||||
model: "gpt-4o-transcribe",
|
||||
silenceDurationMs: 900,
|
||||
vadThreshold: 0.45,
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(resolved).toEqual({
|
||||
model: "gpt-4o-transcribe",
|
||||
silenceDurationMs: 900,
|
||||
vadThreshold: 0.45,
|
||||
});
|
||||
});
|
||||
|
||||
it("accepts the legacy openai-realtime alias", () => {
|
||||
const provider = buildOpenAIRealtimeTranscriptionProvider();
|
||||
expect(provider.aliases).toContain("openai-realtime");
|
||||
});
|
||||
});
|
||||
317
openclaw/extensions/openai/realtime-transcription-provider.ts
Normal file
317
openclaw/extensions/openai/realtime-transcription-provider.ts
Normal file
|
|
@ -0,0 +1,317 @@
|
|||
import { randomUUID } from "node:crypto";
|
||||
import {
|
||||
captureWsEvent,
|
||||
createDebugProxyWebSocketAgent,
|
||||
resolveDebugProxySettings,
|
||||
} from "openclaw/plugin-sdk/proxy-capture";
|
||||
import type {
|
||||
RealtimeTranscriptionProviderConfig,
|
||||
RealtimeTranscriptionProviderPlugin,
|
||||
RealtimeTranscriptionSession,
|
||||
RealtimeTranscriptionSessionCreateRequest,
|
||||
} from "openclaw/plugin-sdk/realtime-transcription";
|
||||
import { normalizeResolvedSecretInputString } from "openclaw/plugin-sdk/secret-input";
|
||||
import WebSocket from "ws";
|
||||
import {
|
||||
asFiniteNumber,
|
||||
readRealtimeErrorDetail,
|
||||
resolveOpenAIProviderConfigRecord,
|
||||
trimToUndefined,
|
||||
} from "./realtime-provider-shared.js";
|
||||
|
||||
type OpenAIRealtimeTranscriptionProviderConfig = {
|
||||
apiKey?: string;
|
||||
model?: string;
|
||||
silenceDurationMs?: number;
|
||||
vadThreshold?: number;
|
||||
};
|
||||
|
||||
type OpenAIRealtimeTranscriptionSessionConfig = RealtimeTranscriptionSessionCreateRequest & {
|
||||
apiKey: string;
|
||||
model: string;
|
||||
silenceDurationMs: number;
|
||||
vadThreshold: number;
|
||||
};
|
||||
|
||||
type RealtimeEvent = {
|
||||
type: string;
|
||||
delta?: string;
|
||||
transcript?: string;
|
||||
error?: unknown;
|
||||
};
|
||||
|
||||
function normalizeProviderConfig(
|
||||
config: RealtimeTranscriptionProviderConfig,
|
||||
): OpenAIRealtimeTranscriptionProviderConfig {
|
||||
const raw = resolveOpenAIProviderConfigRecord(config);
|
||||
return {
|
||||
apiKey:
|
||||
normalizeResolvedSecretInputString({
|
||||
value: raw?.apiKey,
|
||||
path: "plugins.entries.voice-call.config.streaming.providers.openai.apiKey",
|
||||
}) ??
|
||||
normalizeResolvedSecretInputString({
|
||||
value: raw?.openaiApiKey,
|
||||
path: "plugins.entries.voice-call.config.streaming.openaiApiKey",
|
||||
}),
|
||||
model: trimToUndefined(raw?.model) ?? trimToUndefined(raw?.sttModel),
|
||||
silenceDurationMs: asFiniteNumber(raw?.silenceDurationMs),
|
||||
vadThreshold: asFiniteNumber(raw?.vadThreshold),
|
||||
};
|
||||
}
|
||||
|
||||
class OpenAIRealtimeTranscriptionSession implements RealtimeTranscriptionSession {
|
||||
private static readonly MAX_RECONNECT_ATTEMPTS = 5;
|
||||
private static readonly RECONNECT_DELAY_MS = 1000;
|
||||
private static readonly CONNECT_TIMEOUT_MS = 10_000;
|
||||
|
||||
private ws: WebSocket | null = null;
|
||||
private connected = false;
|
||||
private closed = false;
|
||||
private reconnectAttempts = 0;
|
||||
private pendingTranscript = "";
|
||||
private readonly flowId = randomUUID();
|
||||
|
||||
constructor(private readonly config: OpenAIRealtimeTranscriptionSessionConfig) {}
|
||||
|
||||
async connect(): Promise<void> {
|
||||
this.closed = false;
|
||||
this.reconnectAttempts = 0;
|
||||
await this.doConnect();
|
||||
}
|
||||
|
||||
sendAudio(audio: Buffer): void {
|
||||
if (this.ws?.readyState !== WebSocket.OPEN) {
|
||||
return;
|
||||
}
|
||||
this.sendEvent({
|
||||
type: "input_audio_buffer.append",
|
||||
audio: audio.toString("base64"),
|
||||
});
|
||||
}
|
||||
|
||||
close(): void {
|
||||
this.closed = true;
|
||||
this.connected = false;
|
||||
if (this.ws) {
|
||||
this.ws.close(1000, "Transcription session closed");
|
||||
this.ws = null;
|
||||
}
|
||||
}
|
||||
|
||||
isConnected(): boolean {
|
||||
return this.connected;
|
||||
}
|
||||
|
||||
private async doConnect(): Promise<void> {
|
||||
await new Promise<void>((resolve, reject) => {
|
||||
const url = "wss://api.openai.com/v1/realtime?intent=transcription";
|
||||
const debugProxy = resolveDebugProxySettings();
|
||||
const proxyAgent = createDebugProxyWebSocketAgent(debugProxy);
|
||||
this.ws = new WebSocket(url, {
|
||||
headers: {
|
||||
Authorization: `Bearer ${this.config.apiKey}`,
|
||||
"OpenAI-Beta": "realtime=v1",
|
||||
},
|
||||
...(proxyAgent ? { agent: proxyAgent } : {}),
|
||||
});
|
||||
|
||||
const connectTimeout = setTimeout(() => {
|
||||
reject(new Error("OpenAI realtime transcription connection timeout"));
|
||||
}, OpenAIRealtimeTranscriptionSession.CONNECT_TIMEOUT_MS);
|
||||
|
||||
this.ws.on("open", () => {
|
||||
clearTimeout(connectTimeout);
|
||||
this.connected = true;
|
||||
this.reconnectAttempts = 0;
|
||||
captureWsEvent({
|
||||
url,
|
||||
direction: "local",
|
||||
kind: "ws-open",
|
||||
flowId: this.flowId,
|
||||
meta: {
|
||||
provider: "openai",
|
||||
capability: "realtime-transcription",
|
||||
},
|
||||
});
|
||||
this.sendEvent({
|
||||
type: "transcription_session.update",
|
||||
session: {
|
||||
input_audio_format: "g711_ulaw",
|
||||
input_audio_transcription: {
|
||||
model: this.config.model,
|
||||
},
|
||||
turn_detection: {
|
||||
type: "server_vad",
|
||||
threshold: this.config.vadThreshold,
|
||||
prefix_padding_ms: 300,
|
||||
silence_duration_ms: this.config.silenceDurationMs,
|
||||
},
|
||||
},
|
||||
});
|
||||
resolve();
|
||||
});
|
||||
|
||||
this.ws.on("message", (data: Buffer) => {
|
||||
captureWsEvent({
|
||||
url,
|
||||
direction: "inbound",
|
||||
kind: "ws-frame",
|
||||
flowId: this.flowId,
|
||||
payload: data,
|
||||
meta: {
|
||||
provider: "openai",
|
||||
capability: "realtime-transcription",
|
||||
},
|
||||
});
|
||||
try {
|
||||
this.handleEvent(JSON.parse(data.toString()) as RealtimeEvent);
|
||||
} catch (error) {
|
||||
this.config.onError?.(error instanceof Error ? error : new Error(String(error)));
|
||||
}
|
||||
});
|
||||
|
||||
this.ws.on("error", (error) => {
|
||||
captureWsEvent({
|
||||
url,
|
||||
direction: "local",
|
||||
kind: "error",
|
||||
flowId: this.flowId,
|
||||
errorText: error instanceof Error ? error.message : String(error),
|
||||
meta: {
|
||||
provider: "openai",
|
||||
capability: "realtime-transcription",
|
||||
},
|
||||
});
|
||||
if (!this.connected) {
|
||||
clearTimeout(connectTimeout);
|
||||
reject(error);
|
||||
return;
|
||||
}
|
||||
this.config.onError?.(error instanceof Error ? error : new Error(String(error)));
|
||||
});
|
||||
|
||||
this.ws.on("close", (code, reasonBuffer) => {
|
||||
captureWsEvent({
|
||||
url,
|
||||
direction: "local",
|
||||
kind: "ws-close",
|
||||
flowId: this.flowId,
|
||||
closeCode: typeof code === "number" ? code : undefined,
|
||||
meta: {
|
||||
provider: "openai",
|
||||
capability: "realtime-transcription",
|
||||
reason:
|
||||
Buffer.isBuffer(reasonBuffer) && reasonBuffer.length > 0
|
||||
? reasonBuffer.toString("utf8")
|
||||
: undefined,
|
||||
},
|
||||
});
|
||||
this.connected = false;
|
||||
if (this.closed) {
|
||||
return;
|
||||
}
|
||||
void this.attemptReconnect();
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
private async attemptReconnect(): Promise<void> {
|
||||
if (this.closed) {
|
||||
return;
|
||||
}
|
||||
if (this.reconnectAttempts >= OpenAIRealtimeTranscriptionSession.MAX_RECONNECT_ATTEMPTS) {
|
||||
this.config.onError?.(new Error("OpenAI realtime transcription reconnect limit reached"));
|
||||
return;
|
||||
}
|
||||
this.reconnectAttempts += 1;
|
||||
const delay =
|
||||
OpenAIRealtimeTranscriptionSession.RECONNECT_DELAY_MS * 2 ** (this.reconnectAttempts - 1);
|
||||
await new Promise((resolve) => setTimeout(resolve, delay));
|
||||
if (this.closed) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
await this.doConnect();
|
||||
} catch (error) {
|
||||
this.config.onError?.(error instanceof Error ? error : new Error(String(error)));
|
||||
await this.attemptReconnect();
|
||||
}
|
||||
}
|
||||
|
||||
private handleEvent(event: RealtimeEvent): void {
|
||||
switch (event.type) {
|
||||
case "conversation.item.input_audio_transcription.delta":
|
||||
if (event.delta) {
|
||||
this.pendingTranscript += event.delta;
|
||||
this.config.onPartial?.(this.pendingTranscript);
|
||||
}
|
||||
return;
|
||||
|
||||
case "conversation.item.input_audio_transcription.completed":
|
||||
if (event.transcript) {
|
||||
this.config.onTranscript?.(event.transcript);
|
||||
}
|
||||
this.pendingTranscript = "";
|
||||
return;
|
||||
|
||||
case "input_audio_buffer.speech_started":
|
||||
this.pendingTranscript = "";
|
||||
this.config.onSpeechStart?.();
|
||||
return;
|
||||
|
||||
case "error": {
|
||||
const detail = readRealtimeErrorDetail(event.error);
|
||||
this.config.onError?.(new Error(detail));
|
||||
return;
|
||||
}
|
||||
|
||||
default:
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
private sendEvent(event: unknown): void {
|
||||
if (this.ws?.readyState === WebSocket.OPEN) {
|
||||
const payload = JSON.stringify(event);
|
||||
captureWsEvent({
|
||||
url: "wss://api.openai.com/v1/realtime?intent=transcription",
|
||||
direction: "outbound",
|
||||
kind: "ws-frame",
|
||||
flowId: this.flowId,
|
||||
payload,
|
||||
meta: {
|
||||
provider: "openai",
|
||||
capability: "realtime-transcription",
|
||||
},
|
||||
});
|
||||
this.ws.send(payload);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
export function buildOpenAIRealtimeTranscriptionProvider(): RealtimeTranscriptionProviderPlugin {
|
||||
return {
|
||||
id: "openai",
|
||||
label: "OpenAI Realtime Transcription",
|
||||
aliases: ["openai-realtime"],
|
||||
autoSelectOrder: 10,
|
||||
resolveConfig: ({ rawConfig }) => normalizeProviderConfig(rawConfig),
|
||||
isConfigured: ({ providerConfig }) =>
|
||||
Boolean(normalizeProviderConfig(providerConfig).apiKey || process.env.OPENAI_API_KEY),
|
||||
createSession: (req) => {
|
||||
const config = normalizeProviderConfig(req.providerConfig);
|
||||
const apiKey = config.apiKey || process.env.OPENAI_API_KEY;
|
||||
if (!apiKey) {
|
||||
throw new Error("OpenAI API key missing");
|
||||
}
|
||||
return new OpenAIRealtimeTranscriptionSession({
|
||||
...req,
|
||||
apiKey,
|
||||
model: config.model ?? "gpt-4o-transcribe",
|
||||
silenceDurationMs: config.silenceDurationMs ?? 800,
|
||||
vadThreshold: config.vadThreshold ?? 0.5,
|
||||
});
|
||||
},
|
||||
};
|
||||
}
|
||||
30
openclaw/extensions/openai/realtime-voice-provider.test.ts
Normal file
30
openclaw/extensions/openai/realtime-voice-provider.test.ts
Normal file
|
|
@ -0,0 +1,30 @@
|
|||
import { describe, expect, it } from "vitest";
|
||||
import { buildOpenAIRealtimeVoiceProvider } from "./realtime-voice-provider.js";
|
||||
|
||||
describe("buildOpenAIRealtimeVoiceProvider", () => {
|
||||
it("normalizes provider-owned voice settings from raw provider config", () => {
|
||||
const provider = buildOpenAIRealtimeVoiceProvider();
|
||||
const resolved = provider.resolveConfig?.({
|
||||
cfg: {} as never,
|
||||
rawConfig: {
|
||||
providers: {
|
||||
openai: {
|
||||
model: "gpt-realtime",
|
||||
voice: "verse",
|
||||
temperature: 0.6,
|
||||
silenceDurationMs: 850,
|
||||
vadThreshold: 0.35,
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(resolved).toEqual({
|
||||
model: "gpt-realtime",
|
||||
voice: "verse",
|
||||
temperature: 0.6,
|
||||
silenceDurationMs: 850,
|
||||
vadThreshold: 0.35,
|
||||
});
|
||||
});
|
||||
});
|
||||
585
openclaw/extensions/openai/realtime-voice-provider.ts
Normal file
585
openclaw/extensions/openai/realtime-voice-provider.ts
Normal file
|
|
@ -0,0 +1,585 @@
|
|||
import { randomUUID } from "node:crypto";
|
||||
import {
|
||||
captureWsEvent,
|
||||
createDebugProxyWebSocketAgent,
|
||||
resolveDebugProxySettings,
|
||||
} from "openclaw/plugin-sdk/proxy-capture";
|
||||
import type {
|
||||
RealtimeVoiceBridge,
|
||||
RealtimeVoiceBridgeCreateRequest,
|
||||
RealtimeVoiceProviderConfig,
|
||||
RealtimeVoiceProviderPlugin,
|
||||
RealtimeVoiceTool,
|
||||
} from "openclaw/plugin-sdk/realtime-voice";
|
||||
import { normalizeResolvedSecretInputString } from "openclaw/plugin-sdk/secret-input";
|
||||
import WebSocket from "ws";
|
||||
import {
|
||||
asFiniteNumber,
|
||||
readRealtimeErrorDetail,
|
||||
resolveOpenAIProviderConfigRecord,
|
||||
trimToUndefined,
|
||||
} from "./realtime-provider-shared.js";
|
||||
|
||||
export type OpenAIRealtimeVoice =
|
||||
| "alloy"
|
||||
| "ash"
|
||||
| "ballad"
|
||||
| "cedar"
|
||||
| "coral"
|
||||
| "echo"
|
||||
| "marin"
|
||||
| "sage"
|
||||
| "shimmer"
|
||||
| "verse";
|
||||
|
||||
type OpenAIRealtimeVoiceProviderConfig = {
|
||||
apiKey?: string;
|
||||
model?: string;
|
||||
voice?: OpenAIRealtimeVoice;
|
||||
temperature?: number;
|
||||
vadThreshold?: number;
|
||||
silenceDurationMs?: number;
|
||||
prefixPaddingMs?: number;
|
||||
azureEndpoint?: string;
|
||||
azureDeployment?: string;
|
||||
azureApiVersion?: string;
|
||||
};
|
||||
|
||||
type OpenAIRealtimeVoiceBridgeConfig = RealtimeVoiceBridgeCreateRequest & {
|
||||
apiKey: string;
|
||||
model?: string;
|
||||
voice?: OpenAIRealtimeVoice;
|
||||
temperature?: number;
|
||||
vadThreshold?: number;
|
||||
silenceDurationMs?: number;
|
||||
prefixPaddingMs?: number;
|
||||
azureEndpoint?: string;
|
||||
azureDeployment?: string;
|
||||
azureApiVersion?: string;
|
||||
};
|
||||
|
||||
type RealtimeEvent = {
|
||||
type: string;
|
||||
delta?: string;
|
||||
transcript?: string;
|
||||
item_id?: string;
|
||||
call_id?: string;
|
||||
name?: string;
|
||||
error?: unknown;
|
||||
};
|
||||
|
||||
type RealtimeSessionUpdate = {
|
||||
type: "session.update";
|
||||
session: {
|
||||
modalities: string[];
|
||||
instructions?: string;
|
||||
voice: OpenAIRealtimeVoice;
|
||||
input_audio_format: string;
|
||||
output_audio_format: string;
|
||||
turn_detection: {
|
||||
type: "server_vad";
|
||||
threshold: number;
|
||||
prefix_padding_ms: number;
|
||||
silence_duration_ms: number;
|
||||
create_response: boolean;
|
||||
};
|
||||
temperature: number;
|
||||
input_audio_transcription?: { model: string };
|
||||
tools?: RealtimeVoiceTool[];
|
||||
tool_choice?: string;
|
||||
};
|
||||
};
|
||||
|
||||
function normalizeProviderConfig(
|
||||
config: RealtimeVoiceProviderConfig,
|
||||
): OpenAIRealtimeVoiceProviderConfig {
|
||||
const raw = resolveOpenAIProviderConfigRecord(config);
|
||||
return {
|
||||
apiKey: normalizeResolvedSecretInputString({
|
||||
value: raw?.apiKey,
|
||||
path: "plugins.entries.voice-call.config.realtime.providers.openai.apiKey",
|
||||
}),
|
||||
model: trimToUndefined(raw?.model),
|
||||
voice: trimToUndefined(raw?.voice) as OpenAIRealtimeVoice | undefined,
|
||||
temperature: asFiniteNumber(raw?.temperature),
|
||||
vadThreshold: asFiniteNumber(raw?.vadThreshold),
|
||||
silenceDurationMs: asFiniteNumber(raw?.silenceDurationMs),
|
||||
prefixPaddingMs: asFiniteNumber(raw?.prefixPaddingMs),
|
||||
azureEndpoint: trimToUndefined(raw?.azureEndpoint),
|
||||
azureDeployment: trimToUndefined(raw?.azureDeployment),
|
||||
azureApiVersion: trimToUndefined(raw?.azureApiVersion),
|
||||
};
|
||||
}
|
||||
|
||||
function base64ToBuffer(b64: string): Buffer {
|
||||
return Buffer.from(b64, "base64");
|
||||
}
|
||||
|
||||
class OpenAIRealtimeVoiceBridge implements RealtimeVoiceBridge {
|
||||
private static readonly DEFAULT_MODEL = "gpt-realtime";
|
||||
private static readonly MAX_RECONNECT_ATTEMPTS = 5;
|
||||
private static readonly BASE_RECONNECT_DELAY_MS = 1000;
|
||||
private static readonly CONNECT_TIMEOUT_MS = 10_000;
|
||||
|
||||
private ws: WebSocket | null = null;
|
||||
private connected = false;
|
||||
private intentionallyClosed = false;
|
||||
private reconnectAttempts = 0;
|
||||
private pendingAudio: Buffer[] = [];
|
||||
private markQueue: string[] = [];
|
||||
private responseStartTimestamp: number | null = null;
|
||||
private latestMediaTimestamp = 0;
|
||||
private lastAssistantItemId: string | null = null;
|
||||
private toolCallBuffers = new Map<string, { name: string; callId: string; args: string }>();
|
||||
private readonly flowId = randomUUID();
|
||||
|
||||
constructor(private readonly config: OpenAIRealtimeVoiceBridgeConfig) {}
|
||||
|
||||
async connect(): Promise<void> {
|
||||
this.intentionallyClosed = false;
|
||||
this.reconnectAttempts = 0;
|
||||
await this.doConnect();
|
||||
}
|
||||
|
||||
sendAudio(audio: Buffer): void {
|
||||
if (!this.connected || this.ws?.readyState !== WebSocket.OPEN) {
|
||||
if (this.pendingAudio.length < 320) {
|
||||
this.pendingAudio.push(audio);
|
||||
}
|
||||
return;
|
||||
}
|
||||
this.sendEvent({
|
||||
type: "input_audio_buffer.append",
|
||||
audio: audio.toString("base64"),
|
||||
});
|
||||
}
|
||||
|
||||
setMediaTimestamp(ts: number): void {
|
||||
this.latestMediaTimestamp = ts;
|
||||
}
|
||||
|
||||
sendUserMessage(text: string): void {
|
||||
this.sendEvent({
|
||||
type: "conversation.item.create",
|
||||
item: {
|
||||
type: "message",
|
||||
role: "user",
|
||||
content: [{ type: "input_text", text }],
|
||||
},
|
||||
});
|
||||
this.sendEvent({ type: "response.create" });
|
||||
}
|
||||
|
||||
triggerGreeting(instructions?: string): void {
|
||||
if (!this.connected || !this.ws) {
|
||||
return;
|
||||
}
|
||||
this.sendEvent({
|
||||
type: "response.create",
|
||||
response: {
|
||||
instructions: instructions ?? this.config.instructions,
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
submitToolResult(callId: string, result: unknown): void {
|
||||
this.sendEvent({
|
||||
type: "conversation.item.create",
|
||||
item: {
|
||||
type: "function_call_output",
|
||||
call_id: callId,
|
||||
output: JSON.stringify(result),
|
||||
},
|
||||
});
|
||||
this.sendEvent({ type: "response.create" });
|
||||
}
|
||||
|
||||
acknowledgeMark(): void {
|
||||
if (this.markQueue.length === 0) {
|
||||
return;
|
||||
}
|
||||
this.markQueue.shift();
|
||||
if (this.markQueue.length === 0) {
|
||||
this.responseStartTimestamp = null;
|
||||
this.lastAssistantItemId = null;
|
||||
}
|
||||
}
|
||||
|
||||
close(): void {
|
||||
this.intentionallyClosed = true;
|
||||
this.connected = false;
|
||||
if (this.ws) {
|
||||
this.ws.close(1000, "Bridge closed");
|
||||
this.ws = null;
|
||||
}
|
||||
}
|
||||
|
||||
isConnected(): boolean {
|
||||
return this.connected;
|
||||
}
|
||||
|
||||
private async doConnect(): Promise<void> {
|
||||
await new Promise<void>((resolve, reject) => {
|
||||
const { url, headers } = this.resolveConnectionParams();
|
||||
const debugProxy = resolveDebugProxySettings();
|
||||
const proxyAgent = createDebugProxyWebSocketAgent(debugProxy);
|
||||
this.ws = new WebSocket(url, {
|
||||
headers,
|
||||
...(proxyAgent ? { agent: proxyAgent } : {}),
|
||||
});
|
||||
|
||||
const connectTimeout = setTimeout(() => {
|
||||
reject(new Error("OpenAI realtime connection timeout"));
|
||||
}, OpenAIRealtimeVoiceBridge.CONNECT_TIMEOUT_MS);
|
||||
|
||||
this.ws.on("open", () => {
|
||||
clearTimeout(connectTimeout);
|
||||
this.connected = true;
|
||||
this.reconnectAttempts = 0;
|
||||
captureWsEvent({
|
||||
url,
|
||||
direction: "local",
|
||||
kind: "ws-open",
|
||||
flowId: this.flowId,
|
||||
meta: {
|
||||
provider: "openai",
|
||||
capability: "realtime-voice",
|
||||
},
|
||||
});
|
||||
this.sendSessionUpdate();
|
||||
for (const chunk of this.pendingAudio.splice(0)) {
|
||||
this.sendAudio(chunk);
|
||||
}
|
||||
this.config.onReady?.();
|
||||
resolve();
|
||||
});
|
||||
|
||||
this.ws.on("message", (data: Buffer) => {
|
||||
captureWsEvent({
|
||||
url,
|
||||
direction: "inbound",
|
||||
kind: "ws-frame",
|
||||
flowId: this.flowId,
|
||||
payload: data,
|
||||
meta: {
|
||||
provider: "openai",
|
||||
capability: "realtime-voice",
|
||||
},
|
||||
});
|
||||
try {
|
||||
this.handleEvent(JSON.parse(data.toString()) as RealtimeEvent);
|
||||
} catch (error) {
|
||||
console.error("[openai] realtime event parse failed:", error);
|
||||
}
|
||||
});
|
||||
|
||||
this.ws.on("error", (error) => {
|
||||
captureWsEvent({
|
||||
url,
|
||||
direction: "local",
|
||||
kind: "error",
|
||||
flowId: this.flowId,
|
||||
errorText: error instanceof Error ? error.message : String(error),
|
||||
meta: {
|
||||
provider: "openai",
|
||||
capability: "realtime-voice",
|
||||
},
|
||||
});
|
||||
if (!this.connected) {
|
||||
clearTimeout(connectTimeout);
|
||||
reject(error);
|
||||
}
|
||||
this.config.onError?.(error instanceof Error ? error : new Error(String(error)));
|
||||
});
|
||||
|
||||
this.ws.on("close", (code, reasonBuffer) => {
|
||||
captureWsEvent({
|
||||
url,
|
||||
direction: "local",
|
||||
kind: "ws-close",
|
||||
flowId: this.flowId,
|
||||
closeCode: typeof code === "number" ? code : undefined,
|
||||
meta: {
|
||||
provider: "openai",
|
||||
capability: "realtime-voice",
|
||||
reason:
|
||||
Buffer.isBuffer(reasonBuffer) && reasonBuffer.length > 0
|
||||
? reasonBuffer.toString("utf8")
|
||||
: undefined,
|
||||
},
|
||||
});
|
||||
this.connected = false;
|
||||
if (this.intentionallyClosed) {
|
||||
this.config.onClose?.("completed");
|
||||
return;
|
||||
}
|
||||
void this.attemptReconnect();
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
private resolveConnectionParams(): { url: string; headers: Record<string, string> } {
|
||||
const cfg = this.config;
|
||||
if (cfg.azureEndpoint && cfg.azureDeployment) {
|
||||
const base = cfg.azureEndpoint
|
||||
.replace(/\/$/, "")
|
||||
.replace(/^http(s?):/, (_, secure: string) => `ws${secure}:`);
|
||||
const apiVersion = cfg.azureApiVersion ?? "2024-10-01-preview";
|
||||
return {
|
||||
url: `${base}/openai/realtime?api-version=${apiVersion}&deployment=${encodeURIComponent(
|
||||
cfg.azureDeployment,
|
||||
)}`,
|
||||
headers: { "api-key": cfg.apiKey },
|
||||
};
|
||||
}
|
||||
|
||||
if (cfg.azureEndpoint) {
|
||||
const base = cfg.azureEndpoint
|
||||
.replace(/\/$/, "")
|
||||
.replace(/^http(s?):/, (_, secure: string) => `ws${secure}:`);
|
||||
return {
|
||||
url: `${base}/v1/realtime?model=${encodeURIComponent(
|
||||
cfg.model ?? OpenAIRealtimeVoiceBridge.DEFAULT_MODEL,
|
||||
)}`,
|
||||
headers: { Authorization: `Bearer ${cfg.apiKey}` },
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
url: `wss://api.openai.com/v1/realtime?model=${encodeURIComponent(
|
||||
cfg.model ?? OpenAIRealtimeVoiceBridge.DEFAULT_MODEL,
|
||||
)}`,
|
||||
headers: {
|
||||
Authorization: `Bearer ${cfg.apiKey}`,
|
||||
"OpenAI-Beta": "realtime=v1",
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
private async attemptReconnect(): Promise<void> {
|
||||
if (this.intentionallyClosed) {
|
||||
return;
|
||||
}
|
||||
if (this.reconnectAttempts >= OpenAIRealtimeVoiceBridge.MAX_RECONNECT_ATTEMPTS) {
|
||||
this.config.onClose?.("error");
|
||||
return;
|
||||
}
|
||||
this.reconnectAttempts += 1;
|
||||
const delay =
|
||||
OpenAIRealtimeVoiceBridge.BASE_RECONNECT_DELAY_MS * 2 ** (this.reconnectAttempts - 1);
|
||||
await new Promise((resolve) => setTimeout(resolve, delay));
|
||||
if (this.intentionallyClosed) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
await this.doConnect();
|
||||
} catch (error) {
|
||||
this.config.onError?.(error instanceof Error ? error : new Error(String(error)));
|
||||
await this.attemptReconnect();
|
||||
}
|
||||
}
|
||||
|
||||
private sendSessionUpdate(): void {
|
||||
const cfg = this.config;
|
||||
const sessionUpdate: RealtimeSessionUpdate = {
|
||||
type: "session.update",
|
||||
session: {
|
||||
modalities: ["text", "audio"],
|
||||
instructions: cfg.instructions,
|
||||
voice: cfg.voice ?? "alloy",
|
||||
input_audio_format: "g711_ulaw",
|
||||
output_audio_format: "g711_ulaw",
|
||||
input_audio_transcription: {
|
||||
model: "whisper-1",
|
||||
},
|
||||
turn_detection: {
|
||||
type: "server_vad",
|
||||
threshold: cfg.vadThreshold ?? 0.5,
|
||||
prefix_padding_ms: cfg.prefixPaddingMs ?? 300,
|
||||
silence_duration_ms: cfg.silenceDurationMs ?? 500,
|
||||
create_response: true,
|
||||
},
|
||||
temperature: cfg.temperature ?? 0.8,
|
||||
...(cfg.tools && cfg.tools.length > 0
|
||||
? {
|
||||
tools: cfg.tools,
|
||||
tool_choice: "auto",
|
||||
}
|
||||
: {}),
|
||||
},
|
||||
};
|
||||
this.sendEvent(sessionUpdate);
|
||||
}
|
||||
|
||||
private handleEvent(event: RealtimeEvent): void {
|
||||
switch (event.type) {
|
||||
case "response.audio.delta": {
|
||||
if (!event.delta) {
|
||||
return;
|
||||
}
|
||||
const audio = base64ToBuffer(event.delta);
|
||||
this.config.onAudio(audio);
|
||||
if (this.responseStartTimestamp === null) {
|
||||
this.responseStartTimestamp = this.latestMediaTimestamp;
|
||||
}
|
||||
if (event.item_id) {
|
||||
this.lastAssistantItemId = event.item_id;
|
||||
}
|
||||
this.sendMark();
|
||||
return;
|
||||
}
|
||||
|
||||
case "input_audio_buffer.speech_started":
|
||||
this.handleBargeIn();
|
||||
return;
|
||||
|
||||
case "response.audio_transcript.delta":
|
||||
if (event.delta) {
|
||||
this.config.onTranscript?.("assistant", event.delta, false);
|
||||
}
|
||||
return;
|
||||
|
||||
case "response.audio_transcript.done":
|
||||
if (event.transcript) {
|
||||
this.config.onTranscript?.("assistant", event.transcript, true);
|
||||
}
|
||||
return;
|
||||
|
||||
case "conversation.item.input_audio_transcription.completed":
|
||||
if (event.transcript) {
|
||||
this.config.onTranscript?.("user", event.transcript, true);
|
||||
}
|
||||
return;
|
||||
|
||||
case "conversation.item.input_audio_transcription.delta":
|
||||
if (event.delta) {
|
||||
this.config.onTranscript?.("user", event.delta, false);
|
||||
}
|
||||
return;
|
||||
|
||||
case "response.function_call_arguments.delta": {
|
||||
const key = event.item_id ?? "unknown";
|
||||
const existing = this.toolCallBuffers.get(key);
|
||||
if (existing && event.delta) {
|
||||
existing.args += event.delta;
|
||||
} else if (event.item_id) {
|
||||
this.toolCallBuffers.set(event.item_id, {
|
||||
name: event.name ?? "",
|
||||
callId: event.call_id ?? "",
|
||||
args: event.delta ?? "",
|
||||
});
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
case "response.function_call_arguments.done": {
|
||||
const key = event.item_id ?? "unknown";
|
||||
const buffered = this.toolCallBuffers.get(key);
|
||||
if (this.config.onToolCall) {
|
||||
const rawArgs =
|
||||
buffered?.args ||
|
||||
((event as unknown as Record<string, unknown>).arguments as string) ||
|
||||
"{}";
|
||||
let args: unknown = {};
|
||||
try {
|
||||
args = JSON.parse(rawArgs);
|
||||
} catch {}
|
||||
this.config.onToolCall({
|
||||
itemId: key,
|
||||
callId: buffered?.callId || event.call_id || "",
|
||||
name: buffered?.name || event.name || "",
|
||||
args,
|
||||
});
|
||||
}
|
||||
this.toolCallBuffers.delete(key);
|
||||
return;
|
||||
}
|
||||
|
||||
case "error": {
|
||||
const detail = readRealtimeErrorDetail(event.error);
|
||||
this.config.onError?.(new Error(detail));
|
||||
return;
|
||||
}
|
||||
|
||||
default:
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
private handleBargeIn(): void {
|
||||
if (this.markQueue.length > 0 && this.responseStartTimestamp !== null) {
|
||||
const elapsedMs = this.latestMediaTimestamp - this.responseStartTimestamp;
|
||||
if (this.lastAssistantItemId) {
|
||||
this.sendEvent({
|
||||
type: "conversation.item.truncate",
|
||||
item_id: this.lastAssistantItemId,
|
||||
content_index: 0,
|
||||
audio_end_ms: Math.max(0, elapsedMs),
|
||||
});
|
||||
}
|
||||
this.config.onClearAudio();
|
||||
this.markQueue = [];
|
||||
this.lastAssistantItemId = null;
|
||||
this.responseStartTimestamp = null;
|
||||
return;
|
||||
}
|
||||
this.config.onClearAudio();
|
||||
}
|
||||
|
||||
private sendMark(): void {
|
||||
const markName = `audio-${Date.now()}`;
|
||||
this.markQueue.push(markName);
|
||||
this.config.onMark?.(markName);
|
||||
}
|
||||
|
||||
private sendEvent(event: unknown): void {
|
||||
if (this.ws?.readyState === WebSocket.OPEN) {
|
||||
const payload = JSON.stringify(event);
|
||||
captureWsEvent({
|
||||
url: this.resolveConnectionParams().url,
|
||||
direction: "outbound",
|
||||
kind: "ws-frame",
|
||||
flowId: this.flowId,
|
||||
payload,
|
||||
meta: {
|
||||
provider: "openai",
|
||||
capability: "realtime-voice",
|
||||
},
|
||||
});
|
||||
this.ws.send(payload);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
export function buildOpenAIRealtimeVoiceProvider(): RealtimeVoiceProviderPlugin {
|
||||
return {
|
||||
id: "openai",
|
||||
label: "OpenAI Realtime Voice",
|
||||
autoSelectOrder: 10,
|
||||
resolveConfig: ({ rawConfig }) => normalizeProviderConfig(rawConfig),
|
||||
isConfigured: ({ providerConfig }) =>
|
||||
Boolean(normalizeProviderConfig(providerConfig).apiKey || process.env.OPENAI_API_KEY),
|
||||
createBridge: (req) => {
|
||||
const config = normalizeProviderConfig(req.providerConfig);
|
||||
const apiKey = config.apiKey || process.env.OPENAI_API_KEY;
|
||||
if (!apiKey) {
|
||||
throw new Error("OpenAI API key missing");
|
||||
}
|
||||
return new OpenAIRealtimeVoiceBridge({
|
||||
...req,
|
||||
apiKey,
|
||||
model: config.model,
|
||||
voice: config.voice,
|
||||
temperature: config.temperature,
|
||||
vadThreshold: config.vadThreshold,
|
||||
silenceDurationMs: config.silenceDurationMs,
|
||||
prefixPaddingMs: config.prefixPaddingMs,
|
||||
azureEndpoint: config.azureEndpoint,
|
||||
azureDeployment: config.azureDeployment,
|
||||
azureApiVersion: config.azureApiVersion,
|
||||
});
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
export type { OpenAIRealtimeVoiceProviderConfig };
|
||||
16
openclaw/extensions/openai/register.runtime.ts
Normal file
16
openclaw/extensions/openai/register.runtime.ts
Normal file
|
|
@ -0,0 +1,16 @@
|
|||
export { buildOpenAICodexCliBackend } from "./cli-backend.js";
|
||||
export { buildOpenAIImageGenerationProvider } from "./image-generation-provider.js";
|
||||
export {
|
||||
openaiCodexMediaUnderstandingProvider,
|
||||
openaiMediaUnderstandingProvider,
|
||||
} from "./media-understanding-provider.js";
|
||||
export { buildOpenAICodexProviderPlugin } from "./openai-codex-provider.js";
|
||||
export { buildOpenAIProvider } from "./openai-provider.js";
|
||||
export {
|
||||
OPENAI_FRIENDLY_PROMPT_OVERLAY,
|
||||
resolveOpenAIPromptOverlayMode,
|
||||
shouldApplyOpenAIPromptOverlay,
|
||||
} from "./prompt-overlay.js";
|
||||
export { buildOpenAIRealtimeTranscriptionProvider } from "./realtime-transcription-provider.js";
|
||||
export { buildOpenAIRealtimeVoiceProvider } from "./realtime-voice-provider.js";
|
||||
export { buildOpenAISpeechProvider } from "./speech-provider.js";
|
||||
24
openclaw/extensions/openai/replay-policy.ts
Normal file
24
openclaw/extensions/openai/replay-policy.ts
Normal file
|
|
@ -0,0 +1,24 @@
|
|||
import type {
|
||||
ProviderReplayPolicy,
|
||||
ProviderReplayPolicyContext,
|
||||
} from "openclaw/plugin-sdk/plugin-entry";
|
||||
|
||||
/**
|
||||
* Returns the provider-owned replay policy for OpenAI-family transports.
|
||||
*/
|
||||
export function buildOpenAIReplayPolicy(ctx: ProviderReplayPolicyContext): ProviderReplayPolicy {
|
||||
return {
|
||||
sanitizeMode: "images-only",
|
||||
applyAssistantFirstOrderingFix: false,
|
||||
validateGeminiTurns: false,
|
||||
validateAnthropicTurns: false,
|
||||
...(ctx.modelApi === "openai-completions"
|
||||
? {
|
||||
sanitizeToolCallIds: true,
|
||||
toolCallIdMode: "strict" as const,
|
||||
}
|
||||
: {
|
||||
sanitizeToolCallIds: false,
|
||||
}),
|
||||
};
|
||||
}
|
||||
11
openclaw/extensions/openai/setup-api.ts
Normal file
11
openclaw/extensions/openai/setup-api.ts
Normal file
|
|
@ -0,0 +1,11 @@
|
|||
import { definePluginEntry } from "openclaw/plugin-sdk/plugin-entry";
|
||||
import { buildOpenAICodexCliBackend } from "./cli-backend.js";
|
||||
|
||||
export default definePluginEntry({
|
||||
id: "openai",
|
||||
name: "OpenAI Setup",
|
||||
description: "Lightweight OpenAI setup hooks",
|
||||
register(api) {
|
||||
api.registerCliBackend(buildOpenAICodexCliBackend());
|
||||
},
|
||||
});
|
||||
123
openclaw/extensions/openai/shared.ts
Normal file
123
openclaw/extensions/openai/shared.ts
Normal file
|
|
@ -0,0 +1,123 @@
|
|||
import type { OpenClawConfig } from "openclaw/plugin-sdk/config-runtime";
|
||||
import { findCatalogTemplate } from "openclaw/plugin-sdk/provider-catalog-shared";
|
||||
import {
|
||||
cloneFirstTemplateModel,
|
||||
matchesExactOrPrefix,
|
||||
type ProviderPlugin,
|
||||
} from "openclaw/plugin-sdk/provider-model-shared";
|
||||
import { OPENAI_RESPONSES_STREAM_HOOKS } from "openclaw/plugin-sdk/provider-stream-family";
|
||||
import { normalizeOptionalString } from "openclaw/plugin-sdk/text-runtime";
|
||||
import { buildOpenAIReplayPolicy } from "./replay-policy.js";
|
||||
import {
|
||||
resolveOpenAITransportTurnState,
|
||||
resolveOpenAIWebSocketSessionPolicy,
|
||||
} from "./transport-policy.js";
|
||||
|
||||
type SyntheticOpenAIModelCatalogCost = {
|
||||
input: number;
|
||||
output: number;
|
||||
cacheRead: number;
|
||||
cacheWrite: number;
|
||||
};
|
||||
|
||||
type SyntheticOpenAIModelCatalogEntry = {
|
||||
provider: string;
|
||||
id: string;
|
||||
name: string;
|
||||
reasoning?: boolean;
|
||||
input?: ("text" | "image")[];
|
||||
contextWindow?: number;
|
||||
contextTokens?: number;
|
||||
cost?: SyntheticOpenAIModelCatalogCost;
|
||||
};
|
||||
|
||||
export const OPENAI_API_BASE_URL = "https://api.openai.com/v1";
|
||||
|
||||
export function toOpenAIDataUrl(buffer: Buffer, mimeType: string): string {
|
||||
return `data:${mimeType};base64,${buffer.toString("base64")}`;
|
||||
}
|
||||
|
||||
export function resolveConfiguredOpenAIBaseUrl(cfg: OpenClawConfig | undefined): string {
|
||||
return normalizeOptionalString(cfg?.models?.providers?.openai?.baseUrl) ?? OPENAI_API_BASE_URL;
|
||||
}
|
||||
|
||||
function hasSupportedOpenAIResponsesTransport(
|
||||
transport: unknown,
|
||||
): transport is "auto" | "sse" | "websocket" {
|
||||
return transport === "auto" || transport === "sse" || transport === "websocket";
|
||||
}
|
||||
|
||||
export function defaultOpenAIResponsesExtraParams(
|
||||
extraParams: Record<string, unknown> | undefined,
|
||||
options?: { openaiWsWarmup?: boolean },
|
||||
): Record<string, unknown> | undefined {
|
||||
const hasSupportedTransport = hasSupportedOpenAIResponsesTransport(extraParams?.transport);
|
||||
const hasExplicitWarmup = typeof extraParams?.openaiWsWarmup === "boolean";
|
||||
const shouldDefaultWarmup = options?.openaiWsWarmup === true;
|
||||
if (hasSupportedTransport && (!shouldDefaultWarmup || hasExplicitWarmup)) {
|
||||
return extraParams;
|
||||
}
|
||||
|
||||
return {
|
||||
...extraParams,
|
||||
...(hasSupportedTransport ? {} : { transport: "auto" }),
|
||||
...(shouldDefaultWarmup && !hasExplicitWarmup ? { openaiWsWarmup: true } : {}),
|
||||
};
|
||||
}
|
||||
|
||||
type OpenAIResponsesProviderHooks = Pick<
|
||||
ProviderPlugin,
|
||||
| "buildReplayPolicy"
|
||||
| "prepareExtraParams"
|
||||
| "wrapStreamFn"
|
||||
| "resolveTransportTurnState"
|
||||
| "resolveWebSocketSessionPolicy"
|
||||
>;
|
||||
|
||||
const resolveOpenAIResponsesTransportTurnState: NonNullable<
|
||||
OpenAIResponsesProviderHooks["resolveTransportTurnState"]
|
||||
> = (ctx) => resolveOpenAITransportTurnState(ctx);
|
||||
|
||||
const resolveOpenAIResponsesWebSocketSessionPolicy: NonNullable<
|
||||
OpenAIResponsesProviderHooks["resolveWebSocketSessionPolicy"]
|
||||
> = (ctx) => resolveOpenAIWebSocketSessionPolicy(ctx);
|
||||
|
||||
export function buildOpenAIResponsesProviderHooks(options?: {
|
||||
openaiWsWarmup?: boolean;
|
||||
}): OpenAIResponsesProviderHooks {
|
||||
return {
|
||||
buildReplayPolicy: buildOpenAIReplayPolicy,
|
||||
prepareExtraParams: (ctx) => defaultOpenAIResponsesExtraParams(ctx.extraParams, options),
|
||||
...OPENAI_RESPONSES_STREAM_HOOKS,
|
||||
resolveTransportTurnState: resolveOpenAIResponsesTransportTurnState,
|
||||
resolveWebSocketSessionPolicy: resolveOpenAIResponsesWebSocketSessionPolicy,
|
||||
};
|
||||
}
|
||||
|
||||
export function buildOpenAISyntheticCatalogEntry(
|
||||
template: ReturnType<typeof findCatalogTemplate>,
|
||||
entry: {
|
||||
id: string;
|
||||
reasoning: boolean;
|
||||
input: readonly ("text" | "image")[];
|
||||
contextWindow: number;
|
||||
contextTokens?: number;
|
||||
cost?: SyntheticOpenAIModelCatalogCost;
|
||||
},
|
||||
): SyntheticOpenAIModelCatalogEntry | undefined {
|
||||
if (!template) {
|
||||
return undefined;
|
||||
}
|
||||
return {
|
||||
...template,
|
||||
id: entry.id,
|
||||
name: entry.id,
|
||||
reasoning: entry.reasoning,
|
||||
input: [...entry.input],
|
||||
contextWindow: entry.contextWindow,
|
||||
...(entry.contextTokens === undefined ? {} : { contextTokens: entry.contextTokens }),
|
||||
...(entry.cost === undefined ? {} : { cost: entry.cost }),
|
||||
};
|
||||
}
|
||||
|
||||
export { cloneFirstTemplateModel, findCatalogTemplate, matchesExactOrPrefix };
|
||||
195
openclaw/extensions/openai/speech-provider.test.ts
Normal file
195
openclaw/extensions/openai/speech-provider.test.ts
Normal file
|
|
@ -0,0 +1,195 @@
|
|||
import { afterEach, describe, expect, it, vi } from "vitest";
|
||||
import { buildOpenAISpeechProvider } from "./speech-provider.js";
|
||||
|
||||
function isSpeechRequestBody(value: unknown): value is { response_format?: string } {
|
||||
return Boolean(value) && typeof value === "object" && !Array.isArray(value);
|
||||
}
|
||||
|
||||
function parseRequestBody(init: RequestInit | undefined): { response_format?: string } {
|
||||
if (typeof init?.body !== "string") {
|
||||
throw new Error("expected string request body");
|
||||
}
|
||||
const body: unknown = JSON.parse(init.body);
|
||||
if (!isSpeechRequestBody(body)) {
|
||||
throw new Error("expected OpenAI speech request body");
|
||||
}
|
||||
return body;
|
||||
}
|
||||
|
||||
describe("buildOpenAISpeechProvider", () => {
|
||||
const originalFetch = globalThis.fetch;
|
||||
|
||||
afterEach(() => {
|
||||
globalThis.fetch = originalFetch;
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
it("normalizes provider-owned speech config from raw provider config", () => {
|
||||
const provider = buildOpenAISpeechProvider();
|
||||
const resolved = provider.resolveConfig?.({
|
||||
cfg: {} as never,
|
||||
timeoutMs: 30_000,
|
||||
rawConfig: {
|
||||
providers: {
|
||||
openai: {
|
||||
apiKey: "sk-test",
|
||||
baseUrl: "https://example.com/v1/",
|
||||
model: "tts-1",
|
||||
voice: "alloy",
|
||||
speed: 1.25,
|
||||
instructions: " Speak warmly ",
|
||||
responseFormat: " WAV ",
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(resolved).toEqual({
|
||||
apiKey: "sk-test",
|
||||
baseUrl: "https://example.com/v1",
|
||||
model: "tts-1",
|
||||
voice: "alloy",
|
||||
speed: 1.25,
|
||||
instructions: "Speak warmly",
|
||||
responseFormat: "wav",
|
||||
});
|
||||
});
|
||||
|
||||
it("parses OpenAI directive tokens against the resolved base url", () => {
|
||||
const provider = buildOpenAISpeechProvider();
|
||||
|
||||
expect(
|
||||
provider.parseDirectiveToken?.({
|
||||
key: "voice",
|
||||
value: "alloy",
|
||||
policy: {
|
||||
allowVoice: true,
|
||||
allowModelId: true,
|
||||
},
|
||||
providerConfig: {
|
||||
baseUrl: "https://api.openai.com/v1/",
|
||||
},
|
||||
} as never),
|
||||
).toEqual({
|
||||
handled: true,
|
||||
overrides: { voice: "alloy" },
|
||||
});
|
||||
|
||||
expect(
|
||||
provider.parseDirectiveToken?.({
|
||||
key: "model",
|
||||
value: "kokoro-custom-model",
|
||||
policy: {
|
||||
allowVoice: true,
|
||||
allowModelId: true,
|
||||
},
|
||||
providerConfig: {
|
||||
baseUrl: "https://api.openai.com/v1/",
|
||||
},
|
||||
} as never),
|
||||
).toEqual({
|
||||
handled: false,
|
||||
});
|
||||
});
|
||||
|
||||
it("preserves talk responseFormat overrides", () => {
|
||||
const provider = buildOpenAISpeechProvider();
|
||||
|
||||
expect(
|
||||
provider.resolveTalkConfig?.({
|
||||
cfg: {} as never,
|
||||
timeoutMs: 30_000,
|
||||
baseTtsConfig: {
|
||||
providers: {
|
||||
openai: {
|
||||
apiKey: "sk-base",
|
||||
responseFormat: "mp3",
|
||||
},
|
||||
},
|
||||
},
|
||||
talkProviderConfig: {
|
||||
apiKey: "sk-talk",
|
||||
responseFormat: " WAV ",
|
||||
},
|
||||
}),
|
||||
).toMatchObject({
|
||||
apiKey: "sk-talk",
|
||||
responseFormat: "wav",
|
||||
});
|
||||
});
|
||||
|
||||
it("maps Talk speak params onto OpenAI speech overrides", () => {
|
||||
const provider = buildOpenAISpeechProvider();
|
||||
|
||||
expect(
|
||||
provider.resolveTalkOverrides?.({
|
||||
talkProviderConfig: {},
|
||||
params: {
|
||||
text: "Hello from talk mode.",
|
||||
voiceId: "nova",
|
||||
modelId: "tts-1",
|
||||
speed: 218 / 175,
|
||||
},
|
||||
}),
|
||||
).toEqual({
|
||||
voice: "nova",
|
||||
model: "tts-1",
|
||||
speed: 218 / 175,
|
||||
});
|
||||
});
|
||||
|
||||
it("uses wav for Groq-compatible OpenAI TTS endpoints", async () => {
|
||||
const provider = buildOpenAISpeechProvider();
|
||||
const fetchMock = vi.fn(async (_url: string, init?: RequestInit) => {
|
||||
const body = parseRequestBody(init);
|
||||
expect(body.response_format).toBe("wav");
|
||||
return new Response(new Uint8Array([1, 2, 3]), { status: 200 });
|
||||
});
|
||||
globalThis.fetch = fetchMock as unknown as typeof fetch;
|
||||
|
||||
const result = await provider.synthesize({
|
||||
text: "hello",
|
||||
cfg: {} as never,
|
||||
providerConfig: {
|
||||
apiKey: "sk-test",
|
||||
baseUrl: "https://api.groq.com/openai/v1",
|
||||
model: "canopylabs/orpheus-v1-english",
|
||||
voice: "daniel",
|
||||
},
|
||||
target: "audio-file",
|
||||
timeoutMs: 1_000,
|
||||
});
|
||||
|
||||
expect(result.outputFormat).toBe("wav");
|
||||
expect(result.fileExtension).toBe(".wav");
|
||||
expect(result.voiceCompatible).toBe(false);
|
||||
});
|
||||
|
||||
it("honors explicit responseFormat overrides and clears voice-note compatibility when not opus", async () => {
|
||||
const provider = buildOpenAISpeechProvider();
|
||||
const fetchMock = vi.fn(async (_url: string, init?: RequestInit) => {
|
||||
const body = parseRequestBody(init);
|
||||
expect(body.response_format).toBe("wav");
|
||||
return new Response(new Uint8Array([1, 2, 3]), { status: 200 });
|
||||
});
|
||||
globalThis.fetch = fetchMock as unknown as typeof fetch;
|
||||
|
||||
const result = await provider.synthesize({
|
||||
text: "hello",
|
||||
cfg: {} as never,
|
||||
providerConfig: {
|
||||
apiKey: "sk-test",
|
||||
baseUrl: "https://proxy.example.com/openai/v1",
|
||||
model: "canopylabs/orpheus-v1-english",
|
||||
voice: "daniel",
|
||||
responseFormat: "wav",
|
||||
},
|
||||
target: "voice-note",
|
||||
timeoutMs: 1_000,
|
||||
});
|
||||
|
||||
expect(result.outputFormat).toBe("wav");
|
||||
expect(result.fileExtension).toBe(".wav");
|
||||
expect(result.voiceCompatible).toBe(false);
|
||||
});
|
||||
});
|
||||
284
openclaw/extensions/openai/speech-provider.ts
Normal file
284
openclaw/extensions/openai/speech-provider.ts
Normal file
|
|
@ -0,0 +1,284 @@
|
|||
import { normalizeResolvedSecretInputString } from "openclaw/plugin-sdk/secret-input";
|
||||
import type {
|
||||
SpeechDirectiveTokenParseContext,
|
||||
SpeechProviderConfig,
|
||||
SpeechProviderOverrides,
|
||||
SpeechProviderPlugin,
|
||||
} from "openclaw/plugin-sdk/speech";
|
||||
import {
|
||||
normalizeLowercaseStringOrEmpty,
|
||||
normalizeOptionalLowercaseString,
|
||||
} from "openclaw/plugin-sdk/text-runtime";
|
||||
import {
|
||||
asFiniteNumber,
|
||||
asObjectRecord,
|
||||
resolveOpenAIProviderConfigRecord,
|
||||
trimToUndefined,
|
||||
} from "./realtime-provider-shared.js";
|
||||
import {
|
||||
DEFAULT_OPENAI_BASE_URL,
|
||||
isValidOpenAIModel,
|
||||
isValidOpenAIVoice,
|
||||
normalizeOpenAITtsBaseUrl,
|
||||
OPENAI_TTS_MODELS,
|
||||
OPENAI_TTS_VOICES,
|
||||
openaiTTS,
|
||||
} from "./tts.js";
|
||||
|
||||
const OPENAI_SPEECH_RESPONSE_FORMATS = ["mp3", "opus", "wav"] as const;
|
||||
|
||||
type OpenAiSpeechResponseFormat = (typeof OPENAI_SPEECH_RESPONSE_FORMATS)[number];
|
||||
|
||||
type OpenAITtsProviderConfig = {
|
||||
apiKey?: string;
|
||||
baseUrl: string;
|
||||
model: string;
|
||||
voice: string;
|
||||
speed?: number;
|
||||
instructions?: string;
|
||||
responseFormat?: OpenAiSpeechResponseFormat;
|
||||
};
|
||||
|
||||
type OpenAITtsProviderOverrides = {
|
||||
model?: string;
|
||||
voice?: string;
|
||||
speed?: number;
|
||||
};
|
||||
|
||||
function normalizeOpenAISpeechResponseFormat(
|
||||
value: unknown,
|
||||
): OpenAiSpeechResponseFormat | undefined {
|
||||
const next = normalizeOptionalLowercaseString(value);
|
||||
if (!next) {
|
||||
return undefined;
|
||||
}
|
||||
if (
|
||||
OPENAI_SPEECH_RESPONSE_FORMATS.includes(next as (typeof OPENAI_SPEECH_RESPONSE_FORMATS)[number])
|
||||
) {
|
||||
return next as OpenAiSpeechResponseFormat;
|
||||
}
|
||||
throw new Error(`Invalid OpenAI speech responseFormat: ${next}`);
|
||||
}
|
||||
|
||||
function isGroqSpeechBaseUrl(baseUrl: string): boolean {
|
||||
try {
|
||||
const hostname = normalizeLowercaseStringOrEmpty(new URL(baseUrl).hostname);
|
||||
return hostname === "groq.com" || hostname.endsWith(".groq.com");
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
function resolveSpeechResponseFormat(
|
||||
baseUrl: string,
|
||||
target: "audio-file" | "voice-note",
|
||||
configuredFormat?: OpenAiSpeechResponseFormat,
|
||||
): OpenAiSpeechResponseFormat {
|
||||
if (configuredFormat) {
|
||||
return configuredFormat;
|
||||
}
|
||||
if (isGroqSpeechBaseUrl(baseUrl)) {
|
||||
return "wav";
|
||||
}
|
||||
return target === "voice-note" ? "opus" : "mp3";
|
||||
}
|
||||
|
||||
function responseFormatToFileExtension(
|
||||
format: OpenAiSpeechResponseFormat,
|
||||
): ".mp3" | ".opus" | ".wav" {
|
||||
switch (format) {
|
||||
case "opus":
|
||||
return ".opus";
|
||||
case "wav":
|
||||
return ".wav";
|
||||
default:
|
||||
return ".mp3";
|
||||
}
|
||||
}
|
||||
|
||||
function normalizeOpenAIProviderConfig(
|
||||
rawConfig: Record<string, unknown>,
|
||||
): OpenAITtsProviderConfig {
|
||||
const raw = resolveOpenAIProviderConfigRecord(rawConfig);
|
||||
return {
|
||||
apiKey: normalizeResolvedSecretInputString({
|
||||
value: raw?.apiKey,
|
||||
path: "messages.tts.providers.openai.apiKey",
|
||||
}),
|
||||
baseUrl: normalizeOpenAITtsBaseUrl(
|
||||
trimToUndefined(raw?.baseUrl) ??
|
||||
trimToUndefined(process.env.OPENAI_TTS_BASE_URL) ??
|
||||
DEFAULT_OPENAI_BASE_URL,
|
||||
),
|
||||
model: trimToUndefined(raw?.model) ?? "gpt-4o-mini-tts",
|
||||
voice: trimToUndefined(raw?.voice) ?? "coral",
|
||||
speed: asFiniteNumber(raw?.speed),
|
||||
instructions: trimToUndefined(raw?.instructions),
|
||||
responseFormat: normalizeOpenAISpeechResponseFormat(raw?.responseFormat),
|
||||
};
|
||||
}
|
||||
|
||||
function readOpenAIProviderConfig(config: SpeechProviderConfig): OpenAITtsProviderConfig {
|
||||
const normalized = normalizeOpenAIProviderConfig({});
|
||||
return {
|
||||
apiKey: trimToUndefined(config.apiKey) ?? normalized.apiKey,
|
||||
baseUrl: trimToUndefined(config.baseUrl) ?? normalized.baseUrl,
|
||||
model: trimToUndefined(config.model) ?? normalized.model,
|
||||
voice: trimToUndefined(config.voice) ?? normalized.voice,
|
||||
speed: asFiniteNumber(config.speed) ?? normalized.speed,
|
||||
instructions: trimToUndefined(config.instructions) ?? normalized.instructions,
|
||||
responseFormat:
|
||||
normalizeOpenAISpeechResponseFormat(config.responseFormat) ?? normalized.responseFormat,
|
||||
};
|
||||
}
|
||||
|
||||
function readOpenAIOverrides(
|
||||
overrides: SpeechProviderOverrides | undefined,
|
||||
): OpenAITtsProviderOverrides {
|
||||
if (!overrides) {
|
||||
return {};
|
||||
}
|
||||
return {
|
||||
model: trimToUndefined(overrides.model),
|
||||
voice: trimToUndefined(overrides.voice),
|
||||
speed: asFiniteNumber(overrides.speed),
|
||||
};
|
||||
}
|
||||
|
||||
function parseDirectiveToken(ctx: SpeechDirectiveTokenParseContext): {
|
||||
handled: boolean;
|
||||
overrides?: SpeechProviderOverrides;
|
||||
warnings?: string[];
|
||||
} {
|
||||
const baseUrl = trimToUndefined(asObjectRecord(ctx.providerConfig)?.baseUrl);
|
||||
switch (ctx.key) {
|
||||
case "voice":
|
||||
case "openai_voice":
|
||||
case "openaivoice":
|
||||
if (!ctx.policy.allowVoice) {
|
||||
return { handled: true };
|
||||
}
|
||||
if (!isValidOpenAIVoice(ctx.value, baseUrl)) {
|
||||
return { handled: true, warnings: [`invalid OpenAI voice "${ctx.value}"`] };
|
||||
}
|
||||
return { handled: true, overrides: { voice: ctx.value } };
|
||||
case "model":
|
||||
case "openai_model":
|
||||
case "openaimodel":
|
||||
if (!ctx.policy.allowModelId) {
|
||||
return { handled: true };
|
||||
}
|
||||
if (!isValidOpenAIModel(ctx.value, baseUrl)) {
|
||||
return { handled: false };
|
||||
}
|
||||
return { handled: true, overrides: { model: ctx.value } };
|
||||
default:
|
||||
return { handled: false };
|
||||
}
|
||||
}
|
||||
|
||||
export function buildOpenAISpeechProvider(): SpeechProviderPlugin {
|
||||
return {
|
||||
id: "openai",
|
||||
label: "OpenAI",
|
||||
autoSelectOrder: 10,
|
||||
models: OPENAI_TTS_MODELS,
|
||||
voices: OPENAI_TTS_VOICES,
|
||||
resolveConfig: ({ rawConfig }) => normalizeOpenAIProviderConfig(rawConfig),
|
||||
parseDirectiveToken,
|
||||
resolveTalkConfig: ({ baseTtsConfig, talkProviderConfig }) => {
|
||||
const base = normalizeOpenAIProviderConfig(baseTtsConfig);
|
||||
const responseFormat = normalizeOpenAISpeechResponseFormat(talkProviderConfig.responseFormat);
|
||||
return {
|
||||
...base,
|
||||
...(talkProviderConfig.apiKey === undefined
|
||||
? {}
|
||||
: {
|
||||
apiKey: normalizeResolvedSecretInputString({
|
||||
value: talkProviderConfig.apiKey,
|
||||
path: "talk.providers.openai.apiKey",
|
||||
}),
|
||||
}),
|
||||
...(trimToUndefined(talkProviderConfig.baseUrl) == null
|
||||
? {}
|
||||
: { baseUrl: trimToUndefined(talkProviderConfig.baseUrl) }),
|
||||
...(trimToUndefined(talkProviderConfig.modelId) == null
|
||||
? {}
|
||||
: { model: trimToUndefined(talkProviderConfig.modelId) }),
|
||||
...(trimToUndefined(talkProviderConfig.voiceId) == null
|
||||
? {}
|
||||
: { voice: trimToUndefined(talkProviderConfig.voiceId) }),
|
||||
...(asFiniteNumber(talkProviderConfig.speed) == null
|
||||
? {}
|
||||
: { speed: asFiniteNumber(talkProviderConfig.speed) }),
|
||||
...(trimToUndefined(talkProviderConfig.instructions) == null
|
||||
? {}
|
||||
: { instructions: trimToUndefined(talkProviderConfig.instructions) }),
|
||||
...(responseFormat == null ? {} : { responseFormat }),
|
||||
};
|
||||
},
|
||||
resolveTalkOverrides: ({ params }) => ({
|
||||
...(trimToUndefined(params.voiceId) == null
|
||||
? {}
|
||||
: { voice: trimToUndefined(params.voiceId) }),
|
||||
...(trimToUndefined(params.modelId) == null
|
||||
? {}
|
||||
: { model: trimToUndefined(params.modelId) }),
|
||||
...(asFiniteNumber(params.speed) == null ? {} : { speed: asFiniteNumber(params.speed) }),
|
||||
}),
|
||||
listVoices: async () => OPENAI_TTS_VOICES.map((voice) => ({ id: voice, name: voice })),
|
||||
isConfigured: ({ providerConfig }) =>
|
||||
Boolean(readOpenAIProviderConfig(providerConfig).apiKey || process.env.OPENAI_API_KEY),
|
||||
synthesize: async (req) => {
|
||||
const config = readOpenAIProviderConfig(req.providerConfig);
|
||||
const overrides = readOpenAIOverrides(req.providerOverrides);
|
||||
const apiKey = config.apiKey || process.env.OPENAI_API_KEY;
|
||||
if (!apiKey) {
|
||||
throw new Error("OpenAI API key missing");
|
||||
}
|
||||
const responseFormat = resolveSpeechResponseFormat(
|
||||
config.baseUrl,
|
||||
req.target,
|
||||
config.responseFormat,
|
||||
);
|
||||
const audioBuffer = await openaiTTS({
|
||||
text: req.text,
|
||||
apiKey,
|
||||
baseUrl: config.baseUrl,
|
||||
model: overrides.model ?? config.model,
|
||||
voice: overrides.voice ?? config.voice,
|
||||
speed: overrides.speed ?? config.speed,
|
||||
instructions: config.instructions,
|
||||
responseFormat,
|
||||
timeoutMs: req.timeoutMs,
|
||||
});
|
||||
return {
|
||||
audioBuffer,
|
||||
outputFormat: responseFormat,
|
||||
fileExtension: responseFormatToFileExtension(responseFormat),
|
||||
voiceCompatible: req.target === "voice-note" && responseFormat === "opus",
|
||||
};
|
||||
},
|
||||
synthesizeTelephony: async (req) => {
|
||||
const config = readOpenAIProviderConfig(req.providerConfig);
|
||||
const apiKey = config.apiKey || process.env.OPENAI_API_KEY;
|
||||
if (!apiKey) {
|
||||
throw new Error("OpenAI API key missing");
|
||||
}
|
||||
const outputFormat = "pcm";
|
||||
const sampleRate = 24_000;
|
||||
const audioBuffer = await openaiTTS({
|
||||
text: req.text,
|
||||
apiKey,
|
||||
baseUrl: config.baseUrl,
|
||||
model: config.model,
|
||||
voice: config.voice,
|
||||
speed: config.speed,
|
||||
instructions: config.instructions,
|
||||
responseFormat: outputFormat,
|
||||
timeoutMs: req.timeoutMs,
|
||||
});
|
||||
return { audioBuffer, outputFormat, sampleRate };
|
||||
},
|
||||
};
|
||||
}
|
||||
10
openclaw/extensions/openai/test-api.ts
Normal file
10
openclaw/extensions/openai/test-api.ts
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
export { buildOpenAICodexCliBackend } from "./cli-backend.js";
|
||||
export { buildOpenAIImageGenerationProvider } from "./image-generation-provider.js";
|
||||
export {
|
||||
openaiCodexMediaUnderstandingProvider,
|
||||
openaiMediaUnderstandingProvider,
|
||||
} from "./media-understanding-provider.js";
|
||||
export { buildOpenAIRealtimeTranscriptionProvider } from "./realtime-transcription-provider.js";
|
||||
export { buildOpenAIRealtimeVoiceProvider } from "./realtime-voice-provider.js";
|
||||
export { buildOpenAISpeechProvider } from "./speech-provider.js";
|
||||
export { buildOpenAIVideoGenerationProvider } from "./video-generation-provider.js";
|
||||
|
|
@ -0,0 +1,123 @@
|
|||
import { beforeEach, describe, it, vi } from "vitest";
|
||||
import {
|
||||
expectAugmentedCodexCatalog,
|
||||
expectCodexBuiltInSuppression,
|
||||
expectCodexMissingAuthHint,
|
||||
importProviderRuntimeCatalogModule,
|
||||
loadBundledPluginPublicSurfaceSync,
|
||||
} from "../../../test/helpers/plugins/provider-catalog.js";
|
||||
import type { ProviderPlugin } from "../../../test/helpers/plugins/provider-catalog.js";
|
||||
import {
|
||||
registerProviderPlugin,
|
||||
requireRegisteredProvider,
|
||||
} from "../../../test/helpers/plugins/provider-registration.js";
|
||||
|
||||
const PROVIDER_CATALOG_CONTRACT_TIMEOUT_MS = 300_000;
|
||||
|
||||
type ResolvePluginProviders = (params?: { onlyPluginIds?: string[] }) => ProviderPlugin[];
|
||||
type ResolveOwningPluginIdsForProvider = (params: { provider: string }) => string[] | undefined;
|
||||
type ResolveCatalogHookProviderPluginIds = (params: unknown) => string[];
|
||||
|
||||
const resolvePluginProvidersMock = vi.hoisted(() => vi.fn<ResolvePluginProviders>(() => []));
|
||||
const resolveOwningPluginIdsForProviderMock = vi.hoisted(() =>
|
||||
vi.fn<ResolveOwningPluginIdsForProvider>(() => undefined),
|
||||
);
|
||||
const resolveCatalogHookProviderPluginIdsMock = vi.hoisted(() =>
|
||||
vi.fn<ResolveCatalogHookProviderPluginIds>((_) => [] as string[]),
|
||||
);
|
||||
|
||||
vi.mock("../../../src/plugins/providers.js", () => ({
|
||||
resolveOwningPluginIdsForProvider: (params: unknown) =>
|
||||
resolveOwningPluginIdsForProviderMock(params as never),
|
||||
resolveCatalogHookProviderPluginIds: (params: unknown) =>
|
||||
resolveCatalogHookProviderPluginIdsMock(params as never),
|
||||
}));
|
||||
|
||||
vi.mock("../../../src/plugins/providers.runtime.js", () => ({
|
||||
isPluginProvidersLoadInFlight: () => false,
|
||||
resolvePluginProviders: (params: unknown) => resolvePluginProvidersMock(params as never),
|
||||
}));
|
||||
|
||||
export function describeOpenAIProviderCatalogContract() {
|
||||
const contractDepsPromise = (async () => {
|
||||
vi.resetModules();
|
||||
const openaiPlugin = loadBundledPluginPublicSurfaceSync<{
|
||||
default: Parameters<typeof registerProviderPlugin>[0]["plugin"];
|
||||
}>({
|
||||
pluginId: "openai",
|
||||
artifactBasename: "index.js",
|
||||
});
|
||||
const openaiProviders = (
|
||||
await registerProviderPlugin({
|
||||
plugin: openaiPlugin.default,
|
||||
id: "openai",
|
||||
name: "OpenAI",
|
||||
})
|
||||
).providers;
|
||||
const openaiProvider = requireRegisteredProvider(openaiProviders, "openai", "provider");
|
||||
const {
|
||||
augmentModelCatalogWithProviderPlugins,
|
||||
resetProviderRuntimeHookCacheForTest,
|
||||
resolveProviderBuiltInModelSuppression,
|
||||
} = await importProviderRuntimeCatalogModule();
|
||||
return {
|
||||
augmentModelCatalogWithProviderPlugins,
|
||||
resetProviderRuntimeHookCacheForTest,
|
||||
resolveProviderBuiltInModelSuppression,
|
||||
openaiProviders,
|
||||
openaiProvider,
|
||||
};
|
||||
})();
|
||||
|
||||
describe(
|
||||
"openai provider catalog contract",
|
||||
{ timeout: PROVIDER_CATALOG_CONTRACT_TIMEOUT_MS },
|
||||
() => {
|
||||
beforeEach(async () => {
|
||||
const { resetProviderRuntimeHookCacheForTest, openaiProviders } = await contractDepsPromise;
|
||||
resetProviderRuntimeHookCacheForTest();
|
||||
|
||||
resolvePluginProvidersMock.mockReset();
|
||||
resolvePluginProvidersMock.mockImplementation((params?: { onlyPluginIds?: string[] }) => {
|
||||
const onlyPluginIds = params?.onlyPluginIds;
|
||||
if (!onlyPluginIds || onlyPluginIds.length === 0) {
|
||||
return openaiProviders;
|
||||
}
|
||||
return onlyPluginIds.includes("openai") ? openaiProviders : [];
|
||||
});
|
||||
|
||||
resolveOwningPluginIdsForProviderMock.mockReset();
|
||||
resolveOwningPluginIdsForProviderMock.mockImplementation((params) => {
|
||||
switch (params.provider) {
|
||||
case "azure-openai-responses":
|
||||
case "openai":
|
||||
case "openai-codex":
|
||||
return ["openai"];
|
||||
default:
|
||||
return undefined;
|
||||
}
|
||||
});
|
||||
|
||||
resolveCatalogHookProviderPluginIdsMock.mockReset();
|
||||
resolveCatalogHookProviderPluginIdsMock.mockReturnValue(["openai"]);
|
||||
});
|
||||
|
||||
it("keeps codex-only missing-auth hints wired through the provider runtime", async () => {
|
||||
const { openaiProvider } = await contractDepsPromise;
|
||||
expectCodexMissingAuthHint(
|
||||
(params) => openaiProvider.buildMissingAuthMessage?.(params.context) ?? undefined,
|
||||
);
|
||||
});
|
||||
|
||||
it("keeps built-in model suppression wired through the provider runtime", async () => {
|
||||
const { resolveProviderBuiltInModelSuppression } = await contractDepsPromise;
|
||||
expectCodexBuiltInSuppression(resolveProviderBuiltInModelSuppression);
|
||||
});
|
||||
|
||||
it("keeps bundled model augmentation wired through the provider runtime", async () => {
|
||||
const { augmentModelCatalogWithProviderPlugins } = await contractDepsPromise;
|
||||
await expectAugmentedCodexCatalog(augmentModelCatalogWithProviderPlugins);
|
||||
});
|
||||
},
|
||||
);
|
||||
}
|
||||
129
openclaw/extensions/openai/transport-policy.test.ts
Normal file
129
openclaw/extensions/openai/transport-policy.test.ts
Normal file
|
|
@ -0,0 +1,129 @@
|
|||
import type { ProviderRuntimeModel } from "openclaw/plugin-sdk/plugin-entry";
|
||||
import { describe, expect, it } from "vitest";
|
||||
import {
|
||||
resolveOpenAITransportTurnState,
|
||||
resolveOpenAIWebSocketSessionPolicy,
|
||||
} from "./transport-policy.js";
|
||||
|
||||
describe("openai transport policy", () => {
|
||||
const nativeModel = {
|
||||
id: "gpt-5.4",
|
||||
name: "GPT-5.4",
|
||||
api: "openai-responses",
|
||||
provider: "openai",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200000,
|
||||
maxTokens: 8192,
|
||||
} satisfies ProviderRuntimeModel;
|
||||
|
||||
const proxyModel = {
|
||||
...nativeModel,
|
||||
id: "proxy-model",
|
||||
name: "Proxy Model",
|
||||
baseUrl: "https://proxy.example.com/v1",
|
||||
} satisfies ProviderRuntimeModel;
|
||||
|
||||
it("builds native turn state for direct OpenAI routes", () => {
|
||||
expect(
|
||||
resolveOpenAITransportTurnState({
|
||||
provider: "openai",
|
||||
modelId: nativeModel.id,
|
||||
model: nativeModel,
|
||||
sessionId: "session-123",
|
||||
turnId: "turn-123",
|
||||
attempt: 2,
|
||||
transport: "websocket",
|
||||
}),
|
||||
).toMatchObject({
|
||||
headers: {
|
||||
"x-client-request-id": "session-123",
|
||||
"x-openclaw-session-id": "session-123",
|
||||
"x-openclaw-turn-id": "turn-123",
|
||||
"x-openclaw-turn-attempt": "2",
|
||||
},
|
||||
metadata: {
|
||||
openclaw_session_id: "session-123",
|
||||
openclaw_turn_id: "turn-123",
|
||||
openclaw_turn_attempt: "2",
|
||||
openclaw_transport: "websocket",
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it("skips turn state for proxy-like OpenAI routes", () => {
|
||||
expect(
|
||||
resolveOpenAITransportTurnState({
|
||||
provider: "openai",
|
||||
modelId: proxyModel.id,
|
||||
model: proxyModel,
|
||||
sessionId: "session-123",
|
||||
turnId: "turn-123",
|
||||
attempt: 1,
|
||||
transport: "stream",
|
||||
}),
|
||||
).toBeUndefined();
|
||||
});
|
||||
|
||||
it("returns websocket session headers and cooldown for native routes", () => {
|
||||
expect(
|
||||
resolveOpenAIWebSocketSessionPolicy({
|
||||
provider: "openai",
|
||||
modelId: nativeModel.id,
|
||||
model: nativeModel,
|
||||
sessionId: "session-123",
|
||||
}),
|
||||
).toMatchObject({
|
||||
headers: {
|
||||
"x-client-request-id": "session-123",
|
||||
"x-openclaw-session-id": "session-123",
|
||||
},
|
||||
degradeCooldownMs: 60_000,
|
||||
});
|
||||
});
|
||||
|
||||
it("treats Azure routes as native OpenAI-family transports", () => {
|
||||
expect(
|
||||
resolveOpenAIWebSocketSessionPolicy({
|
||||
provider: "azure-openai-responses",
|
||||
modelId: "gpt-5.4",
|
||||
model: {
|
||||
...nativeModel,
|
||||
provider: "azure-openai-responses",
|
||||
baseUrl: "https://demo.openai.azure.com/openai/v1",
|
||||
},
|
||||
sessionId: "session-123",
|
||||
}),
|
||||
).toMatchObject({
|
||||
headers: {
|
||||
"x-client-request-id": "session-123",
|
||||
"x-openclaw-session-id": "session-123",
|
||||
},
|
||||
degradeCooldownMs: 60_000,
|
||||
});
|
||||
});
|
||||
|
||||
it("treats ChatGPT Codex backend routes as native OpenAI-family transports", () => {
|
||||
expect(
|
||||
resolveOpenAIWebSocketSessionPolicy({
|
||||
provider: "openai-codex",
|
||||
modelId: "gpt-5.4",
|
||||
model: {
|
||||
...nativeModel,
|
||||
provider: "openai-codex",
|
||||
api: "openai-codex-responses",
|
||||
baseUrl: "https://chatgpt.com/backend-api",
|
||||
},
|
||||
sessionId: "session-123",
|
||||
}),
|
||||
).toMatchObject({
|
||||
headers: {
|
||||
"x-client-request-id": "session-123",
|
||||
"x-openclaw-session-id": "session-123",
|
||||
},
|
||||
degradeCooldownMs: 60_000,
|
||||
});
|
||||
});
|
||||
});
|
||||
111
openclaw/extensions/openai/transport-policy.ts
Normal file
111
openclaw/extensions/openai/transport-policy.ts
Normal file
|
|
@ -0,0 +1,111 @@
|
|||
import type {
|
||||
ProviderResolveTransportTurnStateContext,
|
||||
ProviderResolveWebSocketSessionPolicyContext,
|
||||
ProviderTransportTurnState,
|
||||
ProviderWebSocketSessionPolicy,
|
||||
} from "openclaw/plugin-sdk/plugin-entry";
|
||||
import { normalizeProviderId } from "openclaw/plugin-sdk/provider-model-shared";
|
||||
import { normalizeLowercaseStringOrEmpty } from "openclaw/plugin-sdk/text-runtime";
|
||||
import { isOpenAIApiBaseUrl, isOpenAICodexBaseUrl } from "./base-url.js";
|
||||
|
||||
const DEFAULT_OPENAI_WS_DEGRADE_COOLDOWN_MS = 60_000;
|
||||
const AZURE_PROVIDER_IDS = new Set(["azure-openai", "azure-openai-responses"]);
|
||||
const OPENAI_CODEX_PROVIDER_ID = "openai-codex";
|
||||
|
||||
function isAzureOpenAIBaseUrl(baseUrl?: string): boolean {
|
||||
const trimmed = baseUrl?.trim();
|
||||
if (!trimmed) {
|
||||
return false;
|
||||
}
|
||||
try {
|
||||
return normalizeLowercaseStringOrEmpty(new URL(trimmed).hostname).endsWith(".openai.azure.com");
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
function normalizeIdentityValue(value: string, maxLength = 160): string {
|
||||
const trimmed = value.trim().replace(/[\r\n]+/g, " ");
|
||||
return trimmed.length > maxLength ? trimmed.slice(0, maxLength) : trimmed;
|
||||
}
|
||||
|
||||
function usesKnownNativeOpenAIRoute(provider: string, baseUrl?: string): boolean {
|
||||
const normalizedProvider = normalizeProviderId(provider);
|
||||
if (!normalizedProvider) {
|
||||
return false;
|
||||
}
|
||||
if (normalizedProvider === "openai") {
|
||||
return !baseUrl || isOpenAIApiBaseUrl(baseUrl);
|
||||
}
|
||||
if (AZURE_PROVIDER_IDS.has(normalizedProvider)) {
|
||||
return !baseUrl || isAzureOpenAIBaseUrl(baseUrl);
|
||||
}
|
||||
if (normalizedProvider === OPENAI_CODEX_PROVIDER_ID) {
|
||||
return !baseUrl || isOpenAIApiBaseUrl(baseUrl) || isOpenAICodexBaseUrl(baseUrl);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
function resolveSessionHeaders(params: {
|
||||
provider: string;
|
||||
baseUrl?: string;
|
||||
sessionId?: string;
|
||||
}): Record<string, string> | undefined {
|
||||
if (!params.sessionId || !usesKnownNativeOpenAIRoute(params.provider, params.baseUrl)) {
|
||||
return undefined;
|
||||
}
|
||||
const sessionId = normalizeIdentityValue(params.sessionId);
|
||||
if (!sessionId) {
|
||||
return undefined;
|
||||
}
|
||||
return {
|
||||
"x-client-request-id": sessionId,
|
||||
"x-openclaw-session-id": sessionId,
|
||||
};
|
||||
}
|
||||
|
||||
export function resolveOpenAITransportTurnState(
|
||||
ctx: ProviderResolveTransportTurnStateContext,
|
||||
): ProviderTransportTurnState | undefined {
|
||||
const sessionHeaders = resolveSessionHeaders({
|
||||
provider: ctx.provider,
|
||||
baseUrl: ctx.model?.baseUrl,
|
||||
sessionId: ctx.sessionId,
|
||||
});
|
||||
if (!sessionHeaders) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const turnId = normalizeIdentityValue(ctx.turnId);
|
||||
const attempt = String(Math.max(1, ctx.attempt));
|
||||
|
||||
return {
|
||||
headers: {
|
||||
...sessionHeaders,
|
||||
"x-openclaw-turn-id": turnId,
|
||||
"x-openclaw-turn-attempt": attempt,
|
||||
},
|
||||
metadata: {
|
||||
openclaw_session_id: sessionHeaders["x-openclaw-session-id"] ?? "",
|
||||
openclaw_turn_id: turnId,
|
||||
openclaw_turn_attempt: attempt,
|
||||
openclaw_transport: ctx.transport,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
export function resolveOpenAIWebSocketSessionPolicy(
|
||||
ctx: ProviderResolveWebSocketSessionPolicyContext,
|
||||
): ProviderWebSocketSessionPolicy | undefined {
|
||||
if (!usesKnownNativeOpenAIRoute(ctx.provider, ctx.model?.baseUrl)) {
|
||||
return undefined;
|
||||
}
|
||||
return {
|
||||
headers: resolveSessionHeaders({
|
||||
provider: ctx.provider,
|
||||
baseUrl: ctx.model?.baseUrl,
|
||||
sessionId: ctx.sessionId,
|
||||
}),
|
||||
degradeCooldownMs: DEFAULT_OPENAI_WS_DEGRADE_COOLDOWN_MS,
|
||||
};
|
||||
}
|
||||
16
openclaw/extensions/openai/tsconfig.json
Normal file
16
openclaw/extensions/openai/tsconfig.json
Normal file
|
|
@ -0,0 +1,16 @@
|
|||
{
|
||||
"extends": "../tsconfig.package-boundary.base.json",
|
||||
"compilerOptions": {
|
||||
"rootDir": "."
|
||||
},
|
||||
"include": ["./*.ts", "./src/**/*.ts"],
|
||||
"exclude": [
|
||||
"./**/*.test.ts",
|
||||
"./dist/**",
|
||||
"./node_modules/**",
|
||||
"./src/test-support/**",
|
||||
"./src/**/*test-helpers.ts",
|
||||
"./src/**/*test-harness.ts",
|
||||
"./src/**/*test-support.ts"
|
||||
]
|
||||
}
|
||||
302
openclaw/extensions/openai/tts.test.ts
Normal file
302
openclaw/extensions/openai/tts.test.ts
Normal file
|
|
@ -0,0 +1,302 @@
|
|||
import { mkdtempSync } from "node:fs";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
import { afterEach, describe, expect, it, vi } from "vitest";
|
||||
import {
|
||||
isValidOpenAIModel,
|
||||
isValidOpenAIVoice,
|
||||
OPENAI_TTS_MODELS,
|
||||
OPENAI_TTS_VOICES,
|
||||
openaiTTS,
|
||||
resolveOpenAITtsInstructions,
|
||||
} from "./tts.js";
|
||||
|
||||
describe("openai tts", () => {
|
||||
const originalFetch = globalThis.fetch;
|
||||
const proxyEnvKeys = [
|
||||
"OPENCLAW_DEBUG_PROXY_ENABLED",
|
||||
"OPENCLAW_DEBUG_PROXY_DB_PATH",
|
||||
"OPENCLAW_DEBUG_PROXY_BLOB_DIR",
|
||||
"OPENCLAW_DEBUG_PROXY_SESSION_ID",
|
||||
] as const;
|
||||
let priorProxyEnv: Partial<Record<(typeof proxyEnvKeys)[number], string | undefined>> = {};
|
||||
|
||||
afterEach(() => {
|
||||
globalThis.fetch = originalFetch;
|
||||
vi.restoreAllMocks();
|
||||
for (const key of proxyEnvKeys) {
|
||||
const value = priorProxyEnv[key];
|
||||
if (value === undefined) {
|
||||
delete process.env[key];
|
||||
} else {
|
||||
process.env[key] = value;
|
||||
}
|
||||
}
|
||||
priorProxyEnv = {};
|
||||
});
|
||||
|
||||
describe("isValidOpenAIVoice", () => {
|
||||
it("accepts all valid OpenAI voices including newer additions", () => {
|
||||
for (const voice of OPENAI_TTS_VOICES) {
|
||||
expect(isValidOpenAIVoice(voice)).toBe(true);
|
||||
}
|
||||
for (const newerVoice of ["ballad", "cedar", "juniper", "marin", "verse"]) {
|
||||
expect(isValidOpenAIVoice(newerVoice), newerVoice).toBe(true);
|
||||
}
|
||||
});
|
||||
|
||||
it("rejects invalid voice names", () => {
|
||||
expect(isValidOpenAIVoice("invalid")).toBe(false);
|
||||
expect(isValidOpenAIVoice("")).toBe(false);
|
||||
expect(isValidOpenAIVoice("ALLOY")).toBe(false);
|
||||
expect(isValidOpenAIVoice("alloy ")).toBe(false);
|
||||
expect(isValidOpenAIVoice(" alloy")).toBe(false);
|
||||
});
|
||||
|
||||
it("treats the default endpoint with trailing slash as the default endpoint", () => {
|
||||
expect(isValidOpenAIVoice("kokoro-custom-voice", "https://api.openai.com/v1/")).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("isValidOpenAIModel", () => {
|
||||
it("matches the supported model set and rejects unsupported values", () => {
|
||||
expect(OPENAI_TTS_MODELS).toContain("gpt-4o-mini-tts");
|
||||
expect(OPENAI_TTS_MODELS).toContain("tts-1");
|
||||
expect(OPENAI_TTS_MODELS).toContain("tts-1-hd");
|
||||
expect(OPENAI_TTS_MODELS).toHaveLength(3);
|
||||
expect(Array.isArray(OPENAI_TTS_MODELS)).toBe(true);
|
||||
expect(OPENAI_TTS_MODELS.length).toBeGreaterThan(0);
|
||||
const cases = [
|
||||
{ model: "gpt-4o-mini-tts", expected: true },
|
||||
{ model: "tts-1", expected: true },
|
||||
{ model: "tts-1-hd", expected: true },
|
||||
{ model: "invalid", expected: false },
|
||||
{ model: "", expected: false },
|
||||
{ model: "gpt-4", expected: false },
|
||||
] as const;
|
||||
for (const testCase of cases) {
|
||||
expect(isValidOpenAIModel(testCase.model), testCase.model).toBe(testCase.expected);
|
||||
}
|
||||
});
|
||||
|
||||
it("treats the default endpoint with trailing slash as the default endpoint", () => {
|
||||
expect(isValidOpenAIModel("kokoro-custom-model", "https://api.openai.com/v1/")).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("resolveOpenAITtsInstructions", () => {
|
||||
it("keeps instructions only for gpt-4o-mini-tts variants", () => {
|
||||
expect(resolveOpenAITtsInstructions("gpt-4o-mini-tts", " Speak warmly ")).toBe(
|
||||
"Speak warmly",
|
||||
);
|
||||
expect(resolveOpenAITtsInstructions("gpt-4o-mini-tts-2025-12-15", "Speak warmly")).toBe(
|
||||
"Speak warmly",
|
||||
);
|
||||
expect(resolveOpenAITtsInstructions("tts-1", "Speak warmly")).toBeUndefined();
|
||||
expect(resolveOpenAITtsInstructions("tts-1-hd", "Speak warmly")).toBeUndefined();
|
||||
expect(resolveOpenAITtsInstructions("gpt-4o-mini-tts", " ")).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe("openaiTTS diagnostics", () => {
|
||||
function createStreamingErrorResponse(params: {
|
||||
status: number;
|
||||
chunkCount: number;
|
||||
chunkSize: number;
|
||||
byte: number;
|
||||
}): { response: Response; getReadCount: () => number } {
|
||||
let reads = 0;
|
||||
const stream = new ReadableStream<Uint8Array>({
|
||||
pull(controller) {
|
||||
if (reads >= params.chunkCount) {
|
||||
controller.close();
|
||||
return;
|
||||
}
|
||||
reads += 1;
|
||||
controller.enqueue(new Uint8Array(params.chunkSize).fill(params.byte));
|
||||
},
|
||||
});
|
||||
return {
|
||||
response: new Response(stream, { status: params.status }),
|
||||
getReadCount: () => reads,
|
||||
};
|
||||
}
|
||||
|
||||
it("includes parsed provider detail and request id for JSON API errors", async () => {
|
||||
const fetchMock = vi.fn(
|
||||
async () =>
|
||||
new Response(
|
||||
JSON.stringify({
|
||||
error: {
|
||||
message: "Invalid API key",
|
||||
type: "invalid_request_error",
|
||||
code: "invalid_api_key",
|
||||
},
|
||||
}),
|
||||
{
|
||||
status: 401,
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"x-request-id": "req_123",
|
||||
},
|
||||
},
|
||||
),
|
||||
);
|
||||
globalThis.fetch = fetchMock as unknown as typeof fetch;
|
||||
|
||||
await expect(
|
||||
openaiTTS({
|
||||
text: "hello",
|
||||
apiKey: "bad-key",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
model: "gpt-4o-mini-tts",
|
||||
voice: "alloy",
|
||||
responseFormat: "mp3",
|
||||
timeoutMs: 5_000,
|
||||
}),
|
||||
).rejects.toThrow(
|
||||
"OpenAI TTS API error (401): Invalid API key [type=invalid_request_error, code=invalid_api_key] [request_id=req_123]",
|
||||
);
|
||||
});
|
||||
|
||||
it("falls back to raw body text when the error body is non-JSON", async () => {
|
||||
const fetchMock = vi.fn(
|
||||
async () => new Response("temporary upstream outage", { status: 503 }),
|
||||
);
|
||||
globalThis.fetch = fetchMock as unknown as typeof fetch;
|
||||
|
||||
await expect(
|
||||
openaiTTS({
|
||||
text: "hello",
|
||||
apiKey: "test-key",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
model: "gpt-4o-mini-tts",
|
||||
voice: "alloy",
|
||||
responseFormat: "mp3",
|
||||
timeoutMs: 5_000,
|
||||
}),
|
||||
).rejects.toThrow("OpenAI TTS API error (503): temporary upstream outage");
|
||||
});
|
||||
|
||||
it("caps streamed non-JSON error reads instead of consuming full response bodies", async () => {
|
||||
const streamed = createStreamingErrorResponse({
|
||||
status: 503,
|
||||
chunkCount: 200,
|
||||
chunkSize: 1024,
|
||||
byte: 120,
|
||||
});
|
||||
const fetchMock = vi.fn(async () => streamed.response);
|
||||
globalThis.fetch = fetchMock as unknown as typeof fetch;
|
||||
|
||||
await expect(
|
||||
openaiTTS({
|
||||
text: "hello",
|
||||
apiKey: "test-key",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
model: "gpt-4o-mini-tts",
|
||||
voice: "alloy",
|
||||
responseFormat: "mp3",
|
||||
timeoutMs: 5_000,
|
||||
}),
|
||||
).rejects.toThrow("OpenAI TTS API error (503)");
|
||||
|
||||
expect(streamed.getReadCount()).toBeLessThan(200);
|
||||
});
|
||||
|
||||
it("records TTS exchanges in debug proxy capture mode", async () => {
|
||||
const tempDir = mkdtempSync(path.join(os.tmpdir(), "openai-tts-capture-"));
|
||||
priorProxyEnv = Object.fromEntries(
|
||||
proxyEnvKeys.map((key) => [key, process.env[key]]),
|
||||
) as typeof priorProxyEnv;
|
||||
process.env.OPENCLAW_DEBUG_PROXY_ENABLED = "1";
|
||||
process.env.OPENCLAW_DEBUG_PROXY_DB_PATH = path.join(tempDir, "capture.sqlite");
|
||||
process.env.OPENCLAW_DEBUG_PROXY_BLOB_DIR = path.join(tempDir, "blobs");
|
||||
process.env.OPENCLAW_DEBUG_PROXY_SESSION_ID = "tts-session";
|
||||
|
||||
globalThis.fetch = vi
|
||||
.fn()
|
||||
.mockResolvedValue(
|
||||
new Response(Buffer.from("audio-bytes"), { status: 200 }),
|
||||
) as unknown as typeof globalThis.fetch;
|
||||
|
||||
const { getDebugProxyCaptureStore } = await import("../../src/proxy-capture/store.sqlite.js");
|
||||
const store = getDebugProxyCaptureStore(
|
||||
process.env.OPENCLAW_DEBUG_PROXY_DB_PATH,
|
||||
process.env.OPENCLAW_DEBUG_PROXY_BLOB_DIR,
|
||||
);
|
||||
store.upsertSession({
|
||||
id: "tts-session",
|
||||
startedAt: Date.now(),
|
||||
mode: "test",
|
||||
sourceScope: "openclaw",
|
||||
sourceProcess: "openclaw",
|
||||
dbPath: process.env.OPENCLAW_DEBUG_PROXY_DB_PATH,
|
||||
blobDir: process.env.OPENCLAW_DEBUG_PROXY_BLOB_DIR,
|
||||
});
|
||||
|
||||
await openaiTTS({
|
||||
text: "hello",
|
||||
apiKey: "test-key",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
model: "gpt-4o-mini-tts",
|
||||
voice: "alloy",
|
||||
responseFormat: "mp3",
|
||||
timeoutMs: 5_000,
|
||||
});
|
||||
await new Promise((resolve) => setTimeout(resolve, 0));
|
||||
|
||||
const events = store.getSessionEvents("tts-session", 10);
|
||||
expect(
|
||||
events.some((event) => event.kind === "request" && event.host === "api.openai.com"),
|
||||
).toBe(true);
|
||||
expect(
|
||||
events.some((event) => event.kind === "response" && event.host === "api.openai.com"),
|
||||
).toBe(true);
|
||||
});
|
||||
|
||||
it("does not double-capture TTS exchanges when the global fetch patch is installed", async () => {
|
||||
const tempDir = mkdtempSync(path.join(os.tmpdir(), "openai-tts-patched-capture-"));
|
||||
priorProxyEnv = Object.fromEntries(
|
||||
proxyEnvKeys.map((key) => [key, process.env[key]]),
|
||||
) as typeof priorProxyEnv;
|
||||
process.env.OPENCLAW_DEBUG_PROXY_ENABLED = "1";
|
||||
process.env.OPENCLAW_DEBUG_PROXY_DB_PATH = path.join(tempDir, "capture.sqlite");
|
||||
process.env.OPENCLAW_DEBUG_PROXY_BLOB_DIR = path.join(tempDir, "blobs");
|
||||
process.env.OPENCLAW_DEBUG_PROXY_SESSION_ID = "tts-patched-session";
|
||||
|
||||
globalThis.fetch = vi
|
||||
.fn()
|
||||
.mockResolvedValue(
|
||||
new Response(Buffer.from("audio-bytes"), { status: 200 }),
|
||||
) as unknown as typeof globalThis.fetch;
|
||||
|
||||
const runtime = await import("../../src/proxy-capture/runtime.js");
|
||||
const { getDebugProxyCaptureStore } = await import("../../src/proxy-capture/store.sqlite.js");
|
||||
runtime.initializeDebugProxyCapture("test");
|
||||
|
||||
await openaiTTS({
|
||||
text: "hello",
|
||||
apiKey: "test-key",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
model: "gpt-4o-mini-tts",
|
||||
voice: "alloy",
|
||||
responseFormat: "mp3",
|
||||
timeoutMs: 5_000,
|
||||
});
|
||||
await new Promise((resolve) => setTimeout(resolve, 0));
|
||||
runtime.finalizeDebugProxyCapture();
|
||||
|
||||
const store = getDebugProxyCaptureStore(
|
||||
process.env.OPENCLAW_DEBUG_PROXY_DB_PATH,
|
||||
process.env.OPENCLAW_DEBUG_PROXY_BLOB_DIR,
|
||||
);
|
||||
const events = store
|
||||
.getSessionEvents("tts-patched-session", 10)
|
||||
.filter((event) => event.host === "api.openai.com");
|
||||
expect(events).toHaveLength(2);
|
||||
const kinds = events.map((event) => String(event.kind)).toSorted();
|
||||
expect(kinds).toEqual(["request", "response"]);
|
||||
store.close();
|
||||
});
|
||||
});
|
||||
});
|
||||
186
openclaw/extensions/openai/tts.ts
Normal file
186
openclaw/extensions/openai/tts.ts
Normal file
|
|
@ -0,0 +1,186 @@
|
|||
import {
|
||||
captureHttpExchange,
|
||||
isDebugProxyGlobalFetchPatchInstalled,
|
||||
} from "openclaw/plugin-sdk/proxy-capture";
|
||||
import {
|
||||
asObject,
|
||||
readResponseTextLimited,
|
||||
trimToUndefined,
|
||||
truncateErrorDetail,
|
||||
} from "openclaw/plugin-sdk/speech";
|
||||
|
||||
export const DEFAULT_OPENAI_BASE_URL = "https://api.openai.com/v1";
|
||||
|
||||
export const OPENAI_TTS_MODELS = ["gpt-4o-mini-tts", "tts-1", "tts-1-hd"] as const;
|
||||
|
||||
export const OPENAI_TTS_VOICES = [
|
||||
"alloy",
|
||||
"ash",
|
||||
"ballad",
|
||||
"cedar",
|
||||
"coral",
|
||||
"echo",
|
||||
"fable",
|
||||
"juniper",
|
||||
"marin",
|
||||
"onyx",
|
||||
"nova",
|
||||
"sage",
|
||||
"shimmer",
|
||||
"verse",
|
||||
] as const;
|
||||
|
||||
type OpenAiTtsVoice = (typeof OPENAI_TTS_VOICES)[number];
|
||||
|
||||
export function normalizeOpenAITtsBaseUrl(baseUrl?: string): string {
|
||||
const trimmed = baseUrl?.trim();
|
||||
if (!trimmed) {
|
||||
return DEFAULT_OPENAI_BASE_URL;
|
||||
}
|
||||
return trimmed.replace(/\/+$/, "");
|
||||
}
|
||||
|
||||
function isCustomOpenAIEndpoint(baseUrl?: string): boolean {
|
||||
if (baseUrl != null) {
|
||||
return normalizeOpenAITtsBaseUrl(baseUrl) !== DEFAULT_OPENAI_BASE_URL;
|
||||
}
|
||||
return normalizeOpenAITtsBaseUrl(process.env.OPENAI_TTS_BASE_URL) !== DEFAULT_OPENAI_BASE_URL;
|
||||
}
|
||||
|
||||
export function isValidOpenAIModel(model: string, baseUrl?: string): boolean {
|
||||
if (isCustomOpenAIEndpoint(baseUrl)) {
|
||||
return true;
|
||||
}
|
||||
return OPENAI_TTS_MODELS.includes(model as (typeof OPENAI_TTS_MODELS)[number]);
|
||||
}
|
||||
|
||||
export function isValidOpenAIVoice(voice: string, baseUrl?: string): voice is OpenAiTtsVoice {
|
||||
if (isCustomOpenAIEndpoint(baseUrl)) {
|
||||
return true;
|
||||
}
|
||||
return OPENAI_TTS_VOICES.includes(voice as OpenAiTtsVoice);
|
||||
}
|
||||
|
||||
export function resolveOpenAITtsInstructions(
|
||||
model: string,
|
||||
instructions?: string,
|
||||
): string | undefined {
|
||||
const next = instructions?.trim();
|
||||
return next && model.includes("gpt-4o-mini-tts") ? next : undefined;
|
||||
}
|
||||
|
||||
function formatOpenAiErrorPayload(payload: unknown): string | undefined {
|
||||
const root = asObject(payload);
|
||||
const subject = asObject(root?.error) ?? root;
|
||||
if (!subject) {
|
||||
return undefined;
|
||||
}
|
||||
const message =
|
||||
trimToUndefined(subject.message) ??
|
||||
trimToUndefined(subject.detail) ??
|
||||
trimToUndefined(root?.message);
|
||||
const type = trimToUndefined(subject.type);
|
||||
const code = trimToUndefined(subject.code);
|
||||
const metadata = [type ? `type=${type}` : undefined, code ? `code=${code}` : undefined]
|
||||
.filter((value): value is string => Boolean(value))
|
||||
.join(", ");
|
||||
if (message && metadata) {
|
||||
return `${truncateErrorDetail(message)} [${metadata}]`;
|
||||
}
|
||||
if (message) {
|
||||
return truncateErrorDetail(message);
|
||||
}
|
||||
if (metadata) {
|
||||
return `[${metadata}]`;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
async function extractOpenAiErrorDetail(response: Response): Promise<string | undefined> {
|
||||
const rawBody = trimToUndefined(await readResponseTextLimited(response));
|
||||
if (!rawBody) {
|
||||
return undefined;
|
||||
}
|
||||
try {
|
||||
return formatOpenAiErrorPayload(JSON.parse(rawBody)) ?? truncateErrorDetail(rawBody);
|
||||
} catch {
|
||||
return truncateErrorDetail(rawBody);
|
||||
}
|
||||
}
|
||||
|
||||
export async function openaiTTS(params: {
|
||||
text: string;
|
||||
apiKey: string;
|
||||
baseUrl: string;
|
||||
model: string;
|
||||
voice: string;
|
||||
speed?: number;
|
||||
instructions?: string;
|
||||
responseFormat: "mp3" | "opus" | "pcm" | "wav";
|
||||
timeoutMs: number;
|
||||
}): Promise<Buffer> {
|
||||
const { text, apiKey, baseUrl, model, voice, speed, instructions, responseFormat, timeoutMs } =
|
||||
params;
|
||||
const effectiveInstructions = resolveOpenAITtsInstructions(model, instructions);
|
||||
|
||||
if (!isValidOpenAIModel(model, baseUrl)) {
|
||||
throw new Error(`Invalid model: ${model}`);
|
||||
}
|
||||
if (!isValidOpenAIVoice(voice, baseUrl)) {
|
||||
throw new Error(`Invalid voice: ${voice}`);
|
||||
}
|
||||
|
||||
const controller = new AbortController();
|
||||
const timeout = setTimeout(() => controller.abort(), timeoutMs);
|
||||
|
||||
try {
|
||||
const requestHeaders = {
|
||||
Authorization: `Bearer ${apiKey}`,
|
||||
"Content-Type": "application/json",
|
||||
};
|
||||
const requestBody = JSON.stringify({
|
||||
model,
|
||||
input: text,
|
||||
voice,
|
||||
response_format: responseFormat,
|
||||
...(speed != null && { speed }),
|
||||
...(effectiveInstructions != null && { instructions: effectiveInstructions }),
|
||||
});
|
||||
const response = await fetch(`${baseUrl}/audio/speech`, {
|
||||
method: "POST",
|
||||
headers: requestHeaders,
|
||||
body: requestBody,
|
||||
signal: controller.signal,
|
||||
});
|
||||
if (!isDebugProxyGlobalFetchPatchInstalled()) {
|
||||
captureHttpExchange({
|
||||
url: `${baseUrl}/audio/speech`,
|
||||
method: "POST",
|
||||
requestHeaders,
|
||||
requestBody,
|
||||
response,
|
||||
transport: "http",
|
||||
meta: {
|
||||
provider: "openai",
|
||||
capability: "tts",
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
if (!response.ok) {
|
||||
const detail = await extractOpenAiErrorDetail(response);
|
||||
const requestId =
|
||||
trimToUndefined(response.headers.get("x-request-id")) ??
|
||||
trimToUndefined(response.headers.get("request-id"));
|
||||
throw new Error(
|
||||
`OpenAI TTS API error (${response.status})` +
|
||||
(detail ? `: ${detail}` : "") +
|
||||
(requestId ? ` [request_id=${requestId}]` : ""),
|
||||
);
|
||||
}
|
||||
|
||||
return Buffer.from(await response.arrayBuffer());
|
||||
} finally {
|
||||
clearTimeout(timeout);
|
||||
}
|
||||
}
|
||||
247
openclaw/extensions/openai/video-generation-provider.test.ts
Normal file
247
openclaw/extensions/openai/video-generation-provider.test.ts
Normal file
|
|
@ -0,0 +1,247 @@
|
|||
import { beforeAll, describe, expect, it, vi } from "vitest";
|
||||
import { expectExplicitVideoGenerationCapabilities } from "../../test/helpers/media-generation/provider-capability-assertions.js";
|
||||
import {
|
||||
getProviderHttpMocks,
|
||||
installProviderHttpMockCleanup,
|
||||
} from "../../test/helpers/media-generation/provider-http-mocks.js";
|
||||
|
||||
const { postJsonRequestMock, fetchWithTimeoutMock, resolveProviderHttpRequestConfigMock } =
|
||||
getProviderHttpMocks();
|
||||
|
||||
let buildOpenAIVideoGenerationProvider: typeof import("./video-generation-provider.js").buildOpenAIVideoGenerationProvider;
|
||||
|
||||
beforeAll(async () => {
|
||||
({ buildOpenAIVideoGenerationProvider } = await import("./video-generation-provider.js"));
|
||||
});
|
||||
|
||||
installProviderHttpMockCleanup();
|
||||
|
||||
describe("openai video generation provider", () => {
|
||||
it("declares explicit mode capabilities", () => {
|
||||
expectExplicitVideoGenerationCapabilities(buildOpenAIVideoGenerationProvider());
|
||||
});
|
||||
|
||||
it("uses JSON for text-only Sora requests", async () => {
|
||||
postJsonRequestMock.mockResolvedValue({
|
||||
response: {
|
||||
json: async () => ({
|
||||
id: "vid_123",
|
||||
model: "sora-2",
|
||||
status: "queued",
|
||||
}),
|
||||
},
|
||||
release: vi.fn(async () => {}),
|
||||
});
|
||||
fetchWithTimeoutMock
|
||||
.mockResolvedValueOnce({
|
||||
json: async () => ({
|
||||
id: "vid_123",
|
||||
model: "sora-2",
|
||||
status: "completed",
|
||||
seconds: "4",
|
||||
size: "720x1280",
|
||||
}),
|
||||
})
|
||||
.mockResolvedValueOnce({
|
||||
headers: new Headers({ "content-type": "video/mp4" }),
|
||||
arrayBuffer: async () => Buffer.from("mp4-bytes"),
|
||||
});
|
||||
|
||||
const provider = buildOpenAIVideoGenerationProvider();
|
||||
const result = await provider.generateVideo({
|
||||
provider: "openai",
|
||||
model: "sora-2",
|
||||
prompt: "A paper airplane gliding through golden hour light",
|
||||
cfg: {},
|
||||
durationSeconds: 4,
|
||||
});
|
||||
|
||||
expect(postJsonRequestMock).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
url: "https://api.openai.com/v1/videos",
|
||||
}),
|
||||
);
|
||||
expect(fetchWithTimeoutMock).toHaveBeenNthCalledWith(
|
||||
1,
|
||||
"https://api.openai.com/v1/videos/vid_123",
|
||||
expect.objectContaining({ method: "GET" }),
|
||||
120000,
|
||||
fetch,
|
||||
);
|
||||
expect(result.videos).toHaveLength(1);
|
||||
expect(result.videos[0]?.mimeType).toBe("video/mp4");
|
||||
expect(result.metadata).toEqual(
|
||||
expect.objectContaining({
|
||||
videoId: "vid_123",
|
||||
status: "completed",
|
||||
}),
|
||||
);
|
||||
});
|
||||
|
||||
it("uses JSON input_reference.image_url for image-to-video requests", async () => {
|
||||
postJsonRequestMock.mockResolvedValue({
|
||||
response: {
|
||||
json: async () => ({
|
||||
id: "vid_456",
|
||||
model: "sora-2",
|
||||
status: "queued",
|
||||
}),
|
||||
},
|
||||
release: vi.fn(async () => {}),
|
||||
});
|
||||
fetchWithTimeoutMock
|
||||
.mockResolvedValueOnce({
|
||||
json: async () => ({
|
||||
id: "vid_456",
|
||||
model: "sora-2",
|
||||
status: "completed",
|
||||
}),
|
||||
})
|
||||
.mockResolvedValueOnce({
|
||||
headers: new Headers({ "content-type": "video/mp4" }),
|
||||
arrayBuffer: async () => Buffer.from("mp4-bytes"),
|
||||
});
|
||||
|
||||
const provider = buildOpenAIVideoGenerationProvider();
|
||||
await provider.generateVideo({
|
||||
provider: "openai",
|
||||
model: "sora-2",
|
||||
prompt: "Animate this frame",
|
||||
cfg: {},
|
||||
inputImages: [{ buffer: Buffer.from("png-bytes"), mimeType: "image/png" }],
|
||||
});
|
||||
|
||||
expect(postJsonRequestMock).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
url: "https://api.openai.com/v1/videos",
|
||||
body: expect.objectContaining({
|
||||
input_reference: {
|
||||
image_url: "data:image/png;base64,cG5nLWJ5dGVz",
|
||||
},
|
||||
}),
|
||||
}),
|
||||
);
|
||||
expect(fetchWithTimeoutMock).toHaveBeenNthCalledWith(
|
||||
1,
|
||||
"https://api.openai.com/v1/videos/vid_456",
|
||||
expect.objectContaining({
|
||||
method: "GET",
|
||||
}),
|
||||
120000,
|
||||
fetch,
|
||||
);
|
||||
});
|
||||
|
||||
it("honors configured baseUrl for video requests", async () => {
|
||||
postJsonRequestMock.mockResolvedValue({
|
||||
response: {
|
||||
json: async () => ({
|
||||
id: "vid_local",
|
||||
model: "sora-2",
|
||||
status: "queued",
|
||||
}),
|
||||
},
|
||||
release: vi.fn(async () => {}),
|
||||
});
|
||||
fetchWithTimeoutMock
|
||||
.mockResolvedValueOnce({
|
||||
json: async () => ({
|
||||
id: "vid_local",
|
||||
model: "sora-2",
|
||||
status: "completed",
|
||||
}),
|
||||
})
|
||||
.mockResolvedValueOnce({
|
||||
headers: new Headers({ "content-type": "video/mp4" }),
|
||||
arrayBuffer: async () => Buffer.from("mp4-bytes"),
|
||||
});
|
||||
|
||||
const provider = buildOpenAIVideoGenerationProvider();
|
||||
await provider.generateVideo({
|
||||
provider: "openai",
|
||||
model: "sora-2",
|
||||
prompt: "Render via local relay",
|
||||
cfg: {
|
||||
models: {
|
||||
providers: {
|
||||
openai: {
|
||||
baseUrl: "http://127.0.0.1:44080/v1",
|
||||
models: [],
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(resolveProviderHttpRequestConfigMock).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
baseUrl: "http://127.0.0.1:44080/v1",
|
||||
}),
|
||||
);
|
||||
expect(postJsonRequestMock).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
url: "http://127.0.0.1:44080/v1/videos",
|
||||
allowPrivateNetwork: false,
|
||||
}),
|
||||
);
|
||||
});
|
||||
|
||||
it("uses multipart input_reference for video-to-video uploads", async () => {
|
||||
fetchWithTimeoutMock
|
||||
.mockResolvedValueOnce({
|
||||
ok: true,
|
||||
json: async () => ({
|
||||
id: "vid_789",
|
||||
model: "sora-2",
|
||||
status: "queued",
|
||||
}),
|
||||
})
|
||||
.mockResolvedValueOnce({
|
||||
json: async () => ({
|
||||
id: "vid_789",
|
||||
model: "sora-2",
|
||||
status: "completed",
|
||||
}),
|
||||
})
|
||||
.mockResolvedValueOnce({
|
||||
headers: new Headers({ "content-type": "video/mp4" }),
|
||||
arrayBuffer: async () => Buffer.from("mp4-bytes"),
|
||||
});
|
||||
|
||||
const provider = buildOpenAIVideoGenerationProvider();
|
||||
await provider.generateVideo({
|
||||
provider: "openai",
|
||||
model: "sora-2",
|
||||
prompt: "Remix this clip",
|
||||
cfg: {},
|
||||
inputVideos: [{ buffer: Buffer.from("mp4-bytes"), mimeType: "video/mp4" }],
|
||||
});
|
||||
|
||||
expect(postJsonRequestMock).not.toHaveBeenCalled();
|
||||
expect(fetchWithTimeoutMock).toHaveBeenNthCalledWith(
|
||||
1,
|
||||
"https://api.openai.com/v1/videos",
|
||||
expect.objectContaining({
|
||||
method: "POST",
|
||||
body: expect.any(FormData),
|
||||
}),
|
||||
120000,
|
||||
fetch,
|
||||
);
|
||||
});
|
||||
|
||||
it("rejects multiple reference assets", async () => {
|
||||
const provider = buildOpenAIVideoGenerationProvider();
|
||||
|
||||
await expect(
|
||||
provider.generateVideo({
|
||||
provider: "openai",
|
||||
model: "sora-2",
|
||||
prompt: "Animate these",
|
||||
cfg: {},
|
||||
inputImages: [{ buffer: Buffer.from("a"), mimeType: "image/png" }],
|
||||
inputVideos: [{ buffer: Buffer.from("b"), mimeType: "video/mp4" }],
|
||||
}),
|
||||
).rejects.toThrow("OpenAI video generation supports at most one reference image or video.");
|
||||
});
|
||||
});
|
||||
388
openclaw/extensions/openai/video-generation-provider.ts
Normal file
388
openclaw/extensions/openai/video-generation-provider.ts
Normal file
|
|
@ -0,0 +1,388 @@
|
|||
import { isProviderApiKeyConfigured } from "openclaw/plugin-sdk/provider-auth";
|
||||
import { resolveApiKeyForProvider } from "openclaw/plugin-sdk/provider-auth-runtime";
|
||||
import {
|
||||
assertOkOrThrowHttpError,
|
||||
createProviderOperationDeadline,
|
||||
fetchWithTimeout,
|
||||
postJsonRequest,
|
||||
resolveProviderOperationTimeoutMs,
|
||||
resolveProviderHttpRequestConfig,
|
||||
waitProviderOperationPollInterval,
|
||||
} from "openclaw/plugin-sdk/provider-http";
|
||||
import { normalizeOptionalString } from "openclaw/plugin-sdk/text-runtime";
|
||||
import type {
|
||||
GeneratedVideoAsset,
|
||||
VideoGenerationProvider,
|
||||
VideoGenerationRequest,
|
||||
} from "openclaw/plugin-sdk/video-generation";
|
||||
import { resolveConfiguredOpenAIBaseUrl, toOpenAIDataUrl } from "./shared.js";
|
||||
|
||||
const DEFAULT_OPENAI_VIDEO_BASE_URL = "https://api.openai.com/v1";
|
||||
const DEFAULT_OPENAI_VIDEO_MODEL = "sora-2";
|
||||
const DEFAULT_TIMEOUT_MS = 120_000;
|
||||
const POLL_INTERVAL_MS = 2_500;
|
||||
const MAX_POLL_ATTEMPTS = 120;
|
||||
const OPENAI_VIDEO_SECONDS = [4, 8, 12] as const;
|
||||
const OPENAI_VIDEO_SIZES = ["720x1280", "1280x720", "1024x1792", "1792x1024"] as const;
|
||||
|
||||
type OpenAIVideoStatus = "queued" | "in_progress" | "completed" | "failed";
|
||||
|
||||
type OpenAIVideoResponse = {
|
||||
id?: string;
|
||||
model?: string;
|
||||
status?: OpenAIVideoStatus;
|
||||
prompt?: string | null;
|
||||
seconds?: string;
|
||||
size?: string;
|
||||
error?: {
|
||||
code?: string;
|
||||
message?: string;
|
||||
} | null;
|
||||
};
|
||||
|
||||
function toBlobBytes(buffer: Buffer): ArrayBuffer {
|
||||
const arrayBuffer = new ArrayBuffer(buffer.byteLength);
|
||||
new Uint8Array(arrayBuffer).set(buffer);
|
||||
return arrayBuffer;
|
||||
}
|
||||
|
||||
function resolveDurationSeconds(durationSeconds: number | undefined): "4" | "8" | "12" | undefined {
|
||||
if (typeof durationSeconds !== "number" || !Number.isFinite(durationSeconds)) {
|
||||
return undefined;
|
||||
}
|
||||
const rounded = Math.max(OPENAI_VIDEO_SECONDS[0], Math.round(durationSeconds));
|
||||
const nearest = OPENAI_VIDEO_SECONDS.reduce((best, current) =>
|
||||
Math.abs(current - rounded) < Math.abs(best - rounded) ? current : best,
|
||||
);
|
||||
return String(nearest) as "4" | "8" | "12";
|
||||
}
|
||||
|
||||
function resolveSize(params: {
|
||||
size?: string;
|
||||
aspectRatio?: string;
|
||||
resolution?: string;
|
||||
}): (typeof OPENAI_VIDEO_SIZES)[number] | undefined {
|
||||
const explicitSize = normalizeOptionalString(params.size);
|
||||
if (
|
||||
explicitSize &&
|
||||
OPENAI_VIDEO_SIZES.includes(explicitSize as (typeof OPENAI_VIDEO_SIZES)[number])
|
||||
) {
|
||||
return explicitSize as (typeof OPENAI_VIDEO_SIZES)[number];
|
||||
}
|
||||
switch (normalizeOptionalString(params.aspectRatio)) {
|
||||
case "9:16":
|
||||
return "720x1280";
|
||||
case "16:9":
|
||||
return "1280x720";
|
||||
case "4:7":
|
||||
return "1024x1792";
|
||||
case "7:4":
|
||||
return "1792x1024";
|
||||
default:
|
||||
break;
|
||||
}
|
||||
if (params.resolution === "1080P") {
|
||||
return "1792x1024";
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function resolveReferenceAsset(req: VideoGenerationRequest) {
|
||||
const allAssets = [...(req.inputImages ?? []), ...(req.inputVideos ?? [])];
|
||||
if (allAssets.length === 0) {
|
||||
return null;
|
||||
}
|
||||
if (allAssets.length > 1) {
|
||||
throw new Error("OpenAI video generation supports at most one reference image or video.");
|
||||
}
|
||||
const [asset] = allAssets;
|
||||
if (!asset?.buffer) {
|
||||
throw new Error(
|
||||
"OpenAI video generation currently requires local image/video uploads for reference assets.",
|
||||
);
|
||||
}
|
||||
const mimeType =
|
||||
normalizeOptionalString(asset.mimeType) ||
|
||||
((req.inputVideos?.length ?? 0) > 0 ? "video/mp4" : "image/png");
|
||||
const extension = mimeType.includes("video")
|
||||
? "mp4"
|
||||
: mimeType.includes("jpeg")
|
||||
? "jpg"
|
||||
: mimeType.includes("webp")
|
||||
? "webp"
|
||||
: "png";
|
||||
const fileName =
|
||||
normalizeOptionalString(asset.fileName) ||
|
||||
`${(req.inputVideos?.length ?? 0) > 0 ? "reference-video" : "reference-image"}.${extension}`;
|
||||
return new File([toBlobBytes(asset.buffer)], fileName, { type: mimeType });
|
||||
}
|
||||
|
||||
async function pollOpenAIVideo(params: {
|
||||
videoId: string;
|
||||
headers: Headers;
|
||||
timeoutMs?: number;
|
||||
baseUrl: string;
|
||||
fetchFn: typeof fetch;
|
||||
}): Promise<OpenAIVideoResponse> {
|
||||
const deadline = createProviderOperationDeadline({
|
||||
timeoutMs: params.timeoutMs,
|
||||
label: `OpenAI video generation task ${params.videoId}`,
|
||||
});
|
||||
for (let attempt = 0; attempt < MAX_POLL_ATTEMPTS; attempt += 1) {
|
||||
const response = await fetchWithTimeout(
|
||||
`${params.baseUrl}/videos/${params.videoId}`,
|
||||
{
|
||||
method: "GET",
|
||||
headers: params.headers,
|
||||
},
|
||||
resolveProviderOperationTimeoutMs({ deadline, defaultTimeoutMs: DEFAULT_TIMEOUT_MS }),
|
||||
params.fetchFn,
|
||||
);
|
||||
await assertOkOrThrowHttpError(response, "OpenAI video status request failed");
|
||||
const payload = (await response.json()) as OpenAIVideoResponse;
|
||||
if (payload.status === "completed") {
|
||||
return payload;
|
||||
}
|
||||
if (payload.status === "failed") {
|
||||
throw new Error(
|
||||
normalizeOptionalString(payload.error?.message) || "OpenAI video generation failed",
|
||||
);
|
||||
}
|
||||
await waitProviderOperationPollInterval({ deadline, pollIntervalMs: POLL_INTERVAL_MS });
|
||||
}
|
||||
throw new Error(`OpenAI video generation task ${params.videoId} did not finish in time`);
|
||||
}
|
||||
|
||||
async function downloadOpenAIVideo(params: {
|
||||
videoId: string;
|
||||
headers: Headers;
|
||||
timeoutMs?: number;
|
||||
baseUrl: string;
|
||||
fetchFn: typeof fetch;
|
||||
}): Promise<GeneratedVideoAsset> {
|
||||
const url = new URL(`${params.baseUrl}/videos/${params.videoId}/content`);
|
||||
url.searchParams.set("variant", "video");
|
||||
const response = await fetchWithTimeout(
|
||||
url.toString(),
|
||||
{
|
||||
method: "GET",
|
||||
headers: new Headers({
|
||||
...Object.fromEntries(params.headers.entries()),
|
||||
Accept: "application/binary",
|
||||
}),
|
||||
},
|
||||
params.timeoutMs ?? DEFAULT_TIMEOUT_MS,
|
||||
params.fetchFn,
|
||||
);
|
||||
await assertOkOrThrowHttpError(response, "OpenAI video download failed");
|
||||
const mimeType = normalizeOptionalString(response.headers.get("content-type")) ?? "video/mp4";
|
||||
const arrayBuffer = await response.arrayBuffer();
|
||||
return {
|
||||
buffer: Buffer.from(arrayBuffer),
|
||||
mimeType,
|
||||
fileName: `video-1.${mimeType.includes("webm") ? "webm" : "mp4"}`,
|
||||
};
|
||||
}
|
||||
|
||||
export function buildOpenAIVideoGenerationProvider(): VideoGenerationProvider {
|
||||
return {
|
||||
id: "openai",
|
||||
label: "OpenAI",
|
||||
defaultModel: DEFAULT_OPENAI_VIDEO_MODEL,
|
||||
models: [DEFAULT_OPENAI_VIDEO_MODEL, "sora-2-pro"],
|
||||
isConfigured: ({ agentDir }) =>
|
||||
isProviderApiKeyConfigured({
|
||||
provider: "openai",
|
||||
agentDir,
|
||||
}),
|
||||
capabilities: {
|
||||
generate: {
|
||||
maxVideos: 1,
|
||||
maxDurationSeconds: 12,
|
||||
supportedDurationSeconds: OPENAI_VIDEO_SECONDS,
|
||||
supportsSize: true,
|
||||
sizes: OPENAI_VIDEO_SIZES,
|
||||
},
|
||||
imageToVideo: {
|
||||
enabled: true,
|
||||
maxVideos: 1,
|
||||
maxInputImages: 1,
|
||||
maxDurationSeconds: 12,
|
||||
supportedDurationSeconds: OPENAI_VIDEO_SECONDS,
|
||||
supportsSize: true,
|
||||
sizes: OPENAI_VIDEO_SIZES,
|
||||
},
|
||||
videoToVideo: {
|
||||
enabled: true,
|
||||
maxVideos: 1,
|
||||
maxInputVideos: 1,
|
||||
maxDurationSeconds: 12,
|
||||
supportedDurationSeconds: OPENAI_VIDEO_SECONDS,
|
||||
supportsSize: true,
|
||||
sizes: OPENAI_VIDEO_SIZES,
|
||||
},
|
||||
},
|
||||
async generateVideo(req) {
|
||||
const auth = await resolveApiKeyForProvider({
|
||||
provider: "openai",
|
||||
cfg: req.cfg,
|
||||
agentDir: req.agentDir,
|
||||
store: req.authStore,
|
||||
});
|
||||
if (!auth.apiKey) {
|
||||
throw new Error("OpenAI API key missing");
|
||||
}
|
||||
|
||||
const fetchFn = fetch;
|
||||
const deadline = createProviderOperationDeadline({
|
||||
timeoutMs: req.timeoutMs,
|
||||
label: "OpenAI video generation",
|
||||
});
|
||||
const { baseUrl, allowPrivateNetwork, headers, dispatcherPolicy } =
|
||||
resolveProviderHttpRequestConfig({
|
||||
baseUrl: resolveConfiguredOpenAIBaseUrl(req.cfg),
|
||||
defaultBaseUrl: DEFAULT_OPENAI_VIDEO_BASE_URL,
|
||||
allowPrivateNetwork: false,
|
||||
defaultHeaders: {
|
||||
Authorization: `Bearer ${auth.apiKey}`,
|
||||
},
|
||||
provider: "openai",
|
||||
capability: "video",
|
||||
transport: "http",
|
||||
});
|
||||
|
||||
const model = normalizeOptionalString(req.model) ?? DEFAULT_OPENAI_VIDEO_MODEL;
|
||||
const seconds = resolveDurationSeconds(req.durationSeconds);
|
||||
const size = resolveSize({
|
||||
size: req.size,
|
||||
aspectRatio: req.aspectRatio,
|
||||
resolution: req.resolution,
|
||||
});
|
||||
const inputImage = req.inputImages?.[0];
|
||||
const referenceAsset = resolveReferenceAsset(req);
|
||||
const requestUrl = `${baseUrl}/videos`;
|
||||
const requestResult = referenceAsset
|
||||
? inputImage?.buffer
|
||||
? await (() => {
|
||||
const jsonHeaders = new Headers(headers);
|
||||
jsonHeaders.set("Content-Type", "application/json");
|
||||
return postJsonRequest({
|
||||
url: requestUrl,
|
||||
headers: jsonHeaders,
|
||||
body: {
|
||||
prompt: req.prompt,
|
||||
model,
|
||||
...(seconds ? { seconds } : {}),
|
||||
...(size ? { size } : {}),
|
||||
input_reference: {
|
||||
image_url: toOpenAIDataUrl(
|
||||
inputImage.buffer,
|
||||
normalizeOptionalString(inputImage.mimeType) ?? "image/png",
|
||||
),
|
||||
},
|
||||
},
|
||||
timeoutMs: resolveProviderOperationTimeoutMs({
|
||||
deadline,
|
||||
defaultTimeoutMs: DEFAULT_TIMEOUT_MS,
|
||||
}),
|
||||
fetchFn,
|
||||
allowPrivateNetwork,
|
||||
dispatcherPolicy,
|
||||
});
|
||||
})()
|
||||
: await (() => {
|
||||
const form = new FormData();
|
||||
form.set("prompt", req.prompt);
|
||||
form.set("model", model);
|
||||
if (seconds) {
|
||||
form.set("seconds", seconds);
|
||||
}
|
||||
if (size) {
|
||||
form.set("size", size);
|
||||
}
|
||||
form.set("input_reference", referenceAsset);
|
||||
const multipartHeaders = new Headers(headers);
|
||||
multipartHeaders.delete("Content-Type");
|
||||
return fetchWithTimeout(
|
||||
requestUrl,
|
||||
{
|
||||
method: "POST",
|
||||
headers: multipartHeaders,
|
||||
body: form,
|
||||
},
|
||||
resolveProviderOperationTimeoutMs({
|
||||
deadline,
|
||||
defaultTimeoutMs: DEFAULT_TIMEOUT_MS,
|
||||
}),
|
||||
fetchFn,
|
||||
).then((response) => ({
|
||||
response,
|
||||
release: async () => {},
|
||||
}));
|
||||
})()
|
||||
: await (() => {
|
||||
const jsonHeaders = new Headers(headers);
|
||||
jsonHeaders.set("Content-Type", "application/json");
|
||||
return postJsonRequest({
|
||||
url: requestUrl,
|
||||
headers: jsonHeaders,
|
||||
body: {
|
||||
prompt: req.prompt,
|
||||
model,
|
||||
...(seconds ? { seconds } : {}),
|
||||
...(size ? { size } : {}),
|
||||
},
|
||||
timeoutMs: resolveProviderOperationTimeoutMs({
|
||||
deadline,
|
||||
defaultTimeoutMs: DEFAULT_TIMEOUT_MS,
|
||||
}),
|
||||
fetchFn,
|
||||
allowPrivateNetwork,
|
||||
dispatcherPolicy,
|
||||
});
|
||||
})();
|
||||
const { response, release } = requestResult;
|
||||
|
||||
try {
|
||||
await assertOkOrThrowHttpError(response, "OpenAI video generation failed");
|
||||
const submitted = (await response.json()) as OpenAIVideoResponse;
|
||||
const videoId = normalizeOptionalString(submitted.id);
|
||||
if (!videoId) {
|
||||
throw new Error("OpenAI video generation response missing video id");
|
||||
}
|
||||
const completed = await pollOpenAIVideo({
|
||||
videoId,
|
||||
headers,
|
||||
timeoutMs: resolveProviderOperationTimeoutMs({
|
||||
deadline,
|
||||
defaultTimeoutMs: DEFAULT_TIMEOUT_MS,
|
||||
}),
|
||||
baseUrl,
|
||||
fetchFn,
|
||||
});
|
||||
const video = await downloadOpenAIVideo({
|
||||
videoId,
|
||||
headers,
|
||||
timeoutMs: resolveProviderOperationTimeoutMs({
|
||||
deadline,
|
||||
defaultTimeoutMs: DEFAULT_TIMEOUT_MS,
|
||||
}),
|
||||
baseUrl,
|
||||
fetchFn,
|
||||
});
|
||||
return {
|
||||
videos: [video],
|
||||
model: completed.model ?? submitted.model ?? model,
|
||||
metadata: {
|
||||
videoId,
|
||||
status: completed.status,
|
||||
seconds: completed.seconds ?? submitted.seconds,
|
||||
size: completed.size ?? submitted.size,
|
||||
},
|
||||
};
|
||||
} finally {
|
||||
await release();
|
||||
}
|
||||
},
|
||||
};
|
||||
}
|
||||
Loading…
Add table
Add a link
Reference in a new issue