fix(chat): complete streaming for reasoning, tools, and tool calls

- Forward reasoning_content and non-stream reasoning to on_token in
  OpenAI chat completions; Gemini thought parts to on_token.
- Stream bubble: always render session.tool panels; format tool_use_call
  payloads and title as call · tool_name.
- Emit tool_use_call via on_tool_ui before tool execution so WS shows
  invoke cards before tool_use_result.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
oliver 2026-05-10 14:02:14 +08:00
parent 75a3038f33
commit 398d85ba06
4 changed files with 68 additions and 17 deletions

View file

@ -1730,6 +1730,21 @@ function formatToolPanelText(name, payload, options = {}) {
};
const n = String(name || "").trim() || "tool";
const p = payload && typeof payload === "object" ? payload : {};
if (String(p.phase || "") === "call" && (p.tool_name != null || p.arguments !== undefined)) {
const tn = String(p.tool_name || "").trim() || "tool";
let argsStr = "";
try {
argsStr =
p.arguments && typeof p.arguments === "object"
? JSON.stringify(p.arguments, null, 2)
: String(p.arguments ?? "");
} catch (_) {
argsStr = String(p.arguments ?? "");
}
const tid = String(p.tool_call_id || "").trim();
const head = currentLang === "zh" ? `[调用] ${tn}` : `[call] ${tn}`;
return normalizeStreamText([head, tid ? `tool_call_id: ${tid}` : "", "", argsStr].filter(Boolean).join("\n"));
}
const r = p.result && typeof p.result === "object" ? p.result : p;
// SQL audit payload: keep line breaks and key fields visible.
if (r && typeof r === "object" && (r.input_sql || r.executed_sql || r.sql_guard)) {
@ -4370,16 +4385,15 @@ async function renderChatUi() {
const title = String(seg.title || "tool");
const body = String(seg.body || "");
const images = Array.isArray(seg.images) ? seg.images : [];
if (!showToolOutput && !images.length) continue;
// Live stream: always render tool cards from session.tool so过程态完整;showToolOutput 仍用于历史聚合气泡。
flushTextBuf();
const audit = seg.sqlAudit && typeof seg.sqlAudit === "object" ? seg.sqlAudit : null;
if (showToolOutput) {
if (audit) {
const guard = audit.guard && typeof audit.guard === "object" ? audit.guard : {};
const autoLimit = _sqlLimitSuffix(audit.inputSql, audit.executedSql);
const executedDiffHtml = _renderExecutedSqlWithAddedHighlight(audit.inputSql, audit.executedSql);
blocks.push(
`<details class="chat-msg__reasoning"><summary>${escapeHtml(title)} · SQL audit</summary>
if (audit) {
const guard = audit.guard && typeof audit.guard === "object" ? audit.guard : {};
const autoLimit = _sqlLimitSuffix(audit.inputSql, audit.executedSql);
const executedDiffHtml = _renderExecutedSqlWithAddedHighlight(audit.inputSql, audit.executedSql);
blocks.push(
`<details class="chat-msg__reasoning"><summary>${escapeHtml(title)} · SQL audit</summary>
<div style="display:grid;grid-template-columns:1fr 1fr;gap:8px;margin-top:8px;">
<div><div class="muted" style="margin-bottom:4px;">input SQL</div><pre class="chat-msg__reasoning-pre">${escapeHtml(String(audit.inputSql || ""))}</pre></div>
<div><div class="muted" style="margin-bottom:4px;">executed SQL (added highlighted)</div><pre class="chat-msg__reasoning-pre">${executedDiffHtml}</pre></div>
@ -4394,12 +4408,11 @@ ${autoLimit ? `<div style="margin-top:8px;"><span class="muted">auto-added claus
rows_returned=${escapeHtml(String(audit.rowsReturned != null ? audit.rowsReturned : ""))}
</div>
</details>`,
);
} else {
blocks.push(
`<details class="chat-msg__reasoning"><summary>${escapeHtml(title)}</summary><pre class="chat-msg__reasoning-pre">${escapeHtml(body)}</pre></details>`,
);
}
);
} else {
blocks.push(
`<details class="chat-msg__reasoning"><summary>${escapeHtml(title)}</summary><pre class="chat-msg__reasoning-pre">${escapeHtml(body)}</pre></details>`,
);
}
for (const im of images) {
const src = String((im && im.src) || "").trim();
@ -4508,6 +4521,14 @@ ${autoLimit ? `<div style="margin-top:8px;"><span class="muted">auto-added claus
const liveTag = currentLang === "zh" ? "[实时全量]" : "[LIVE FULL]";
const key = `${name}:${++toolSeq}`;
const rawPayload = p.payload != null ? p.payload : p;
const titleTool =
rawPayload && typeof rawPayload === "object" && String(rawPayload.tool_name || "").trim()
? String(rawPayload.tool_name || "").trim()
: "";
const displayName =
name === "tool_use_call" && titleTool
? `${currentLang === "zh" ? "调用" : "call"} · ${titleTool}`
: name;
const body = formatToolPanelText(name, rawPayload, { streamMode: true });
const sqlAudit = extractSqlAuditPayload(rawPayload);
const images = extractToolImageItems(rawPayload);
@ -4536,7 +4557,14 @@ ${autoLimit ? `<div style="margin-top:8px;"><span class="muted">auto-added claus
return false;
};
if (_containsRefAttachments(rawPayload)) sawStreamToolRefAttachments = true;
chatStreamSegments.push({ type: "tool", key, title: `${name} ${liveTag}`, body, sqlAudit, images });
chatStreamSegments.push({
type: "tool",
key,
title: `${displayName} ${liveTag}`.trim(),
body,
sqlAudit,
images,
});
};
const appendFinalAssistant = async (message, fallbackText) => {
const wsAttachments =

View file

@ -224,6 +224,8 @@ class GoogleGeminiChatModel(ChatModel):
txt = str(p.get("text") or "")
if txt:
thinking_parts.append(txt)
if on_token:
on_token(txt)
sig = p.get("thoughtSignature")
if isinstance(sig, str) and sig:
thinking_signature = sig

View file

@ -551,8 +551,11 @@ class OpenAIChatModel(ChatModel):
reasoning_parts = getattr(msg, "reasoning_content", None) or ""
reasoning_text = str(reasoning_parts).strip() if reasoning_parts else ""
content = msg.content or ""
if on_token and content:
on_token(content)
if on_token:
if reasoning_text:
on_token(reasoning_text)
if content:
on_token(content)
tool_calls: list[LLMToolCall] = []
if msg.tool_calls:
@ -620,6 +623,8 @@ class OpenAIChatModel(ChatModel):
rc = getattr(delta, "reasoning_content", None) or ""
if rc:
reasoning_parts.append(rc)
if on_token:
on_token(rc)
if delta.tool_calls:
for tc in delta.tool_calls:
idx = int(tc.index)

View file

@ -1354,6 +1354,22 @@ def run_oclaw_direct_loop(
# force one no-tool synthesis pass to guarantee a visible assistant body.
hit_tool_round_limit = True
# Stream UI: emit one session.tool per pending call before execution (tool_use_result fires after).
if callable(on_tool_ui):
for tc in step.llm_tool_calls:
try:
on_tool_ui(
"tool_use_call",
{
"phase": "call",
"tool_name": str(getattr(tc, "name", "") or ""),
"tool_call_id": str(getattr(tc, "id", "") or ""),
"arguments": dict(getattr(tc, "arguments", {}) or {}),
},
)
except Exception:
pass
elapsed_ms, results_by_id = _execute_tool_step(
skill_exec=skill_exec,
store=store,