diff --git a/interfaces/admin/models_api.py b/interfaces/admin/models_api.py index 7dfb0632..0d4b4f4f 100644 --- a/interfaces/admin/models_api.py +++ b/interfaces/admin/models_api.py @@ -276,6 +276,17 @@ def include_model_mgmt_routes( mode_save = mode_raw model = payload.get("model") base_url = payload.get("base_url") + thinking_mode_enabled = payload.get("thinking_mode_enabled") + if thinking_mode_enabled is None: + think_enabled: bool | None = None + else: + think_enabled = bool(thinking_mode_enabled) + reasoning_effort_raw = payload.get("reasoning_effort") + reasoning_effort: str | None = None + if reasoning_effort_raw is not None: + reasoning_effort = str(reasoning_effort_raw or "").strip().lower() + if reasoning_effort not in ("", "low", "medium", "high"): + raise HTTPException(status_code=400, detail="invalid_reasoning_effort") model_s = str(model).strip() if model is not None else str(prof.get("model") or "").strip() bu_s = str(base_url).strip() if base_url is not None else str(prof.get("base_url") or "").strip() store.update_llm_profile( @@ -284,6 +295,8 @@ def include_model_mgmt_routes( mode=mode_save, model=model_s or None, base_url=bu_s or None, + thinking_mode_enabled=think_enabled, + reasoning_effort=reasoning_effort, ) store.set_setting(_active_key(ctx), pid) return {"ok": True, "profile": store.get_llm_profile(pid)} diff --git a/interfaces/admin/static/theme-deepseek.css b/interfaces/admin/static/theme-deepseek.css index c5ffef6a..ffbba424 100644 --- a/interfaces/admin/static/theme-deepseek.css +++ b/interfaces/admin/static/theme-deepseek.css @@ -471,6 +471,7 @@ body.theme-ds-body .card { /* 与 .chat-avatar-slot 48px + .chat-row gap 10px 一致 */ .chat-msg-col--assistant { + width: min(max(0px, calc(50% + 2cm - 58px)), 100%); max-width: min(max(0px, calc(50% + 2cm - 58px)), 100%); } @@ -570,6 +571,7 @@ body.theme-ds-body .card { .chat-msg--assistant { align-self: flex-start; + width: 100%; background: rgba(255, 255, 255, 0.04); border: 1px solid var(--ds-border, rgba(255, 255, 255, 0.08)); } diff --git a/platform/llm/transports/openai_chat_completions.py b/platform/llm/transports/openai_chat_completions.py index cf5b3e2d..fb5d88df 100644 --- a/platform/llm/transports/openai_chat_completions.py +++ b/platform/llm/transports/openai_chat_completions.py @@ -30,6 +30,47 @@ def _is_minimax_compat(model: str | None, base_url: str | None) -> bool: return ("minimax" in m) or ("minimax" in b) +def _truthy_env(name: str, default: str = "") -> bool: + v = os.getenv(name) + if v is None: + v = default + return str(v or "").strip().lower() in ("1", "true", "yes", "on") + + +def _should_disable_thinking(base_url: str | None) -> bool: + # Some OpenAI-compatible gateways enable "thinking" mode and require replaying + # reasoning_content in subsequent turns. When the client does not preserve it, + # the gateway returns HTTP 400. Allow disabling thinking at request level. + # + # Safety: do NOT send unknown fields to official OpenAI endpoints by default. + if _truthy_env("AIA_LLM_THINKING_FORCE_DISABLED", "0"): + return True + if _truthy_env("AIA_LLM_THINKING_FORCE_ENABLED", "0"): + return False + b = str(base_url or "").strip().lower() + if not b: + return False + if "api.openai.com" in b: + return False + # Default-on for non-official OpenAI-compatible gateways. + return _truthy_env("AIA_LLM_THINKING_DISABLED", "1") + + +def _should_enable_thinking(base_url: str | None, *, thinking_mode_enabled: bool = False) -> bool: + if _truthy_env("AIA_LLM_THINKING_FORCE_ENABLED", "0"): + return True + if _truthy_env("AIA_LLM_THINKING_FORCE_DISABLED", "0"): + return False + if not bool(thinking_mode_enabled): + return False + b = str(base_url or "").strip().lower() + if not b: + return False + if "api.openai.com" in b: + return False + return True + + def _find_thought_signature_in_obj(o: Any) -> str | None: if isinstance(o, dict): for k in ("thought_signature", "thoughtSignature"): @@ -102,10 +143,15 @@ class OpenAIChatModel(ChatModel): model: str | None = None, api_key: str | None = None, base_url: str | None = None, + thinking_mode_enabled: bool = False, + reasoning_effort: str | None = None, ): self.model = model or os.getenv("OPENAI_MODEL") or "gpt-4o-mini" self.api_key = api_key or os.getenv("OPENAI_API_KEY") self.base_url = base_url or os.getenv("OPENAI_BASE_URL") + self.thinking_mode_enabled = bool(thinking_mode_enabled) + eff = str(reasoning_effort or "").strip().lower() + self.reasoning_effort = eff if eff in ("low", "medium", "high") else "" if not self.api_key: raise RuntimeError("未设置 OPENAI_API_KEY,无法使用 OpenAI 模型") @@ -177,6 +223,19 @@ class OpenAIChatModel(ChatModel): cleaned_msgs.append(m) kwargs: dict[str, Any] = {"model": self.model, "messages": cleaned_msgs, "stream": stream} + if _should_enable_thinking(self.base_url, thinking_mode_enabled=bool(getattr(self, "thinking_mode_enabled", False))): + extra_body = kwargs.get("extra_body") if isinstance(kwargs.get("extra_body"), dict) else {} + extra_body = dict(extra_body) + extra_body["thinking"] = {"type": "enabled"} + kwargs["extra_body"] = extra_body + eff = str(getattr(self, "reasoning_effort", "") or "").strip().lower() + if eff in ("low", "medium", "high"): + kwargs["reasoning_effort"] = eff + elif _should_disable_thinking(self.base_url): + extra_body = kwargs.get("extra_body") if isinstance(kwargs.get("extra_body"), dict) else {} + extra_body = dict(extra_body) + extra_body["thinking"] = {"type": "disabled"} + kwargs["extra_body"] = extra_body if use_tools: try: from oclaw.platform.config.paths import db_path @@ -195,7 +254,20 @@ class OpenAIChatModel(ChatModel): kwargs["tools"] = plan.tools_wired except Exception: kwargs["tools"] = tools - return self._client.chat.completions.create(**kwargs) + try: + return self._client.chat.completions.create(**kwargs) + except Exception as exc: + msg = str(exc) + if "reasoning_content" in msg and "thinking mode" in msg and "must be passed back" in msg: + # Provider requires replaying assistant.reasoning_content in thinking mode. + # As a safety fallback, force-disable thinking and retry once. + extra_body = kwargs.get("extra_body") if isinstance(kwargs.get("extra_body"), dict) else {} + extra_body = dict(extra_body) + extra_body["thinking"] = {"type": "disabled"} + kwargs["extra_body"] = extra_body + kwargs.pop("reasoning_effort", None) + return self._client.chat.completions.create(**kwargs) + raise def _llm_response_from_completion(self, completion: Any, *, on_token: Optional[Callable[[str], None]]) -> LLMResponse: msg = completion.choices[0].message diff --git a/platform/llm/transports/openai_responses.py b/platform/llm/transports/openai_responses.py index 2fbb2c40..479171dd 100644 --- a/platform/llm/transports/openai_responses.py +++ b/platform/llm/transports/openai_responses.py @@ -104,10 +104,21 @@ def parse_openai_responses_stream_events( class OpenAIResponsesModel(ChatModel): """OpenAI Responses API transport (OpenAI-compatible gateways may implement this surface).""" - def __init__(self, *, model: str | None = None, api_key: str | None = None, base_url: str | None = None): + def __init__( + self, + *, + model: str | None = None, + api_key: str | None = None, + base_url: str | None = None, + thinking_mode_enabled: bool = False, + reasoning_effort: str | None = None, + ): self.model = (model or os.getenv("OPENAI_MODEL") or "gpt-4o-mini").strip() self.api_key = (api_key or os.getenv("OPENAI_API_KEY") or "").strip() self.base_url = (base_url or os.getenv("OPENAI_BASE_URL") or "").strip() or None + self.thinking_mode_enabled = bool(thinking_mode_enabled) + eff = str(reasoning_effort or "").strip().lower() + self.reasoning_effort = eff if eff in ("low", "medium", "high") else "" if not self.api_key: raise RuntimeError("未设置 OPENAI_API_KEY,无法使用 OpenAI Responses") try: @@ -188,9 +199,41 @@ class OpenAIResponsesModel(ChatModel): norm = self._normalize_messages(messages) # OpenAI-compatible gateways differ: some require input={"messages":[...]} with role=user only. stream_errors: list[str] = [] + b = str(self.base_url or "").strip().lower() + force_disable = str(os.getenv("AIA_LLM_THINKING_FORCE_DISABLED") or "").strip().lower() in ("1", "true", "yes", "on") + force_enable = str(os.getenv("AIA_LLM_THINKING_FORCE_ENABLED") or "").strip().lower() in ("1", "true", "yes", "on") + mode_enabled = bool(getattr(self, "thinking_mode_enabled", False)) + extra_body: dict[str, Any] = {} + if b and ("api.openai.com" not in b): + if force_disable: + extra_body["thinking"] = {"type": "disabled"} + elif force_enable or mode_enabled: + extra_body["thinking"] = {"type": "enabled"} + else: + # Default: disable thinking for non-official gateways unless explicitly enabled. + if str(os.getenv("AIA_LLM_THINKING_DISABLED") or "1").strip().lower() in ("1", "true", "yes", "on"): + extra_body["thinking"] = {"type": "disabled"} + thinking = {"extra_body": extra_body} if extra_body else {} + reasoning_effort = str(getattr(self, "reasoning_effort", "") or "").strip().lower() + if reasoning_effort not in ("low", "medium", "high"): + reasoning_effort = "" stream_variants: list[dict[str, Any]] = [ - {"model": self.model, "input": {"messages": norm}, "tools": tools or None, "stream": True}, - {"model": self.model, "input": norm, "tools": tools or None, "stream": True}, + { + **thinking, + **({"reasoning_effort": reasoning_effort} if reasoning_effort else {}), + "model": self.model, + "input": {"messages": norm}, + "tools": tools or None, + "stream": True, + }, + { + **thinking, + **({"reasoning_effort": reasoning_effort} if reasoning_effort else {}), + "model": self.model, + "input": norm, + "tools": tools or None, + "stream": True, + }, ] try: for payload in stream_variants: @@ -205,15 +248,48 @@ class OpenAIResponsesModel(ChatModel): on_token(text) return LLMResponse(content=text, tool_calls=tool_calls) except Exception as exc: - stream_errors.append(str(exc)) + emsg = str(exc) + # Fallback: provider thinking-mode replay contract. + if "reasoning_content" in emsg and "thinking mode" in emsg and "must be passed back" in emsg: + try: + forced = dict(payload) + eb = forced.get("extra_body") if isinstance(forced.get("extra_body"), dict) else {} + eb = dict(eb) + eb["thinking"] = {"type": "disabled"} + forced["extra_body"] = eb + forced.pop("reasoning_effort", None) + stream = self._client.responses.create(**forced) + text, tool_calls, final_resp = parse_openai_responses_stream_events(stream, on_token=on_token) + if (not text.strip()) and final_resp: + ot = final_resp.get("output_text") + if isinstance(ot, str) and ot.strip(): + text = ot + if on_token: + on_token(text) + return LLMResponse(content=text, tool_calls=tool_calls) + except Exception: + pass + stream_errors.append(emsg) continue raise RuntimeError("; ".join(stream_errors) or "responses_stream_all_variants_failed") except Exception as exc: logger.info("responses stream failed; fallback to non-stream (%s)", exc) nonstream_errors: list[str] = [] for payload in ( - {"model": self.model, "input": {"messages": norm}, "tools": tools or None}, - {"model": self.model, "input": norm, "tools": tools or None}, + { + **thinking, + **({"reasoning_effort": reasoning_effort} if reasoning_effort else {}), + "model": self.model, + "input": {"messages": norm}, + "tools": tools or None, + }, + { + **thinking, + **({"reasoning_effort": reasoning_effort} if reasoning_effort else {}), + "model": self.model, + "input": norm, + "tools": tools or None, + }, ): try: resp = self._client.responses.create(**payload) @@ -224,7 +300,25 @@ class OpenAIResponsesModel(ChatModel): on_token(text) return LLMResponse(content=text, tool_calls=tool_calls) except Exception as e2: - nonstream_errors.append(str(e2)) + emsg2 = str(e2) + if "reasoning_content" in emsg2 and "thinking mode" in emsg2 and "must be passed back" in emsg2: + try: + forced = dict(payload) + eb = forced.get("extra_body") if isinstance(forced.get("extra_body"), dict) else {} + eb = dict(eb) + eb["thinking"] = {"type": "disabled"} + forced["extra_body"] = eb + forced.pop("reasoning_effort", None) + resp = self._client.responses.create(**forced) + d = _as_dict(resp) or {} + text = str(d.get("output_text") or "") + tool_calls = _collect_tool_calls_from_response_dict(d) + if on_token and text: + on_token(text) + return LLMResponse(content=text, tool_calls=tool_calls) + except Exception: + pass + nonstream_errors.append(emsg2) continue raise RuntimeError( "openai_responses_request_failed: " diff --git a/platform/persistence/sqlite_store.py b/platform/persistence/sqlite_store.py index f3b1da8f..eb1673ba 100644 --- a/platform/persistence/sqlite_store.py +++ b/platform/persistence/sqlite_store.py @@ -912,6 +912,10 @@ class SqliteStore: conn.execute("ALTER TABLE llm_profile ADD COLUMN hide_in_ui INTEGER NOT NULL DEFAULT 0") if "owner_user_id" not in prof_cols: conn.execute("ALTER TABLE llm_profile ADD COLUMN owner_user_id TEXT") + if "thinking_mode_enabled" not in prof_cols: + conn.execute("ALTER TABLE llm_profile ADD COLUMN thinking_mode_enabled INTEGER NOT NULL DEFAULT 0") + if "reasoning_effort" not in prof_cols: + conn.execute("ALTER TABLE llm_profile ADD COLUMN reasoning_effort TEXT") conn.execute( """ CREATE TABLE IF NOT EXISTS llm_profile_user_grant ( @@ -2349,8 +2353,8 @@ class SqliteStore: conn.execute( """ INSERT INTO llm_profile - (id, name, mode, model, base_url, api_key, updated_at, is_builtin, hide_in_ui, owner_user_id) - VALUES (?, ?, ?, ?, ?, NULL, ?, 0, 0, ?) + (id, name, mode, model, base_url, api_key, updated_at, is_builtin, hide_in_ui, owner_user_id, thinking_mode_enabled, reasoning_effort) + VALUES (?, ?, ?, ?, ?, NULL, ?, 0, 0, ?, 0, '') """, (profile_id, name, mode, model, base_url, ts, own), ) @@ -2393,7 +2397,9 @@ class SqliteStore: SELECT id, name, mode, model, base_url, api_key, updated_at, COALESCE(is_builtin, 0) AS is_builtin, COALESCE(hide_in_ui, 0) AS hide_in_ui, - owner_user_id + owner_user_id, + COALESCE(thinking_mode_enabled, 0) AS thinking_mode_enabled, + COALESCE(reasoning_effort, '') AS reasoning_effort FROM llm_profile {where_sql} ORDER BY COALESCE(is_builtin, 0) DESC, COALESCE(hide_in_ui, 0) ASC, updated_at DESC @@ -2465,6 +2471,8 @@ class SqliteStore: "is_builtin": is_builtin, "hide_in_ui": bool(int(r["hide_in_ui"] or 0)), "owner_user_id": own, + "thinking_mode_enabled": bool(int(r["thinking_mode_enabled"] or 0)), + "reasoning_effort": str(r["reasoning_effort"] or "").strip().lower(), "mutable": mutable, "visibility_reason": vis, } @@ -2478,7 +2486,9 @@ class SqliteStore: SELECT id, name, mode, model, base_url, api_key, updated_at, COALESCE(is_builtin, 0) AS is_builtin, COALESCE(hide_in_ui, 0) AS hide_in_ui, - owner_user_id + owner_user_id, + COALESCE(thinking_mode_enabled, 0) AS thinking_mode_enabled, + COALESCE(reasoning_effort, '') AS reasoning_effort FROM llm_profile WHERE id = ? """, @@ -2497,6 +2507,8 @@ class SqliteStore: "is_builtin": bool(int(r["is_builtin"] or 0)), "hide_in_ui": bool(int(r["hide_in_ui"] or 0)), "owner_user_id": str(r["owner_user_id"] or "").strip() if r["owner_user_id"] is not None else "", + "thinking_mode_enabled": bool(int(r["thinking_mode_enabled"] or 0)), + "reasoning_effort": str(r["reasoning_effort"] or "").strip().lower(), } def update_llm_profile( @@ -2506,16 +2518,26 @@ class SqliteStore: mode: str, model: str | None, base_url: str | None, + *, + thinking_mode_enabled: bool | None = None, + reasoning_effort: str | None = None, ) -> None: ts = utc_now_iso() + think_val = None if thinking_mode_enabled is None else (1 if bool(thinking_mode_enabled) else 0) + eff = None if reasoning_effort is None else str(reasoning_effort or "").strip().lower() + if eff is not None and eff not in ("", "low", "medium", "high"): + eff = "" with self._connect() as conn: conn.execute( """ UPDATE llm_profile - SET name = ?, mode = ?, model = ?, base_url = ?, updated_at = ? + SET name = ?, mode = ?, model = ?, base_url = ?, + thinking_mode_enabled = COALESCE(?, thinking_mode_enabled), + reasoning_effort = COALESCE(?, reasoning_effort), + updated_at = ? WHERE id = ? """, - (name, mode, model, base_url, ts, profile_id), + (name, mode, model, base_url, think_val, eff, ts, profile_id), ) def delete_llm_profile(self, profile_id: str) -> None: diff --git a/runtime/agents/factory.py b/runtime/agents/factory.py index 5b4a51c9..e6e61382 100644 --- a/runtime/agents/factory.py +++ b/runtime/agents/factory.py @@ -90,6 +90,24 @@ def _build_executor_components( m = (raw or "").strip().lower() return m if m in ("openai", "openai_responses", "anthropic", "ollama", "rule", "google") else "rule" + def _profile_thinking_config(profile: dict[str, Any] | None) -> tuple[bool, str]: + if not isinstance(profile, dict): + return False, "" + think = bool(profile.get("thinking_mode_enabled")) + eff = str(profile.get("reasoning_effort") or "").strip().lower() + if eff not in ("", "low", "medium", "high"): + eff = "" + return think, eff + + def _apply_profile_thinking(model_obj: object, profile: dict[str, Any] | None) -> object: + think, eff = _profile_thinking_config(profile) + try: + setattr(model_obj, "thinking_mode_enabled", think) + setattr(model_obj, "reasoning_effort", eff) + except Exception: + pass + return model_obj + def _build_chat_model_for_profile( target_profile_id: str | None, *, @@ -123,7 +141,7 @@ def _build_executor_components( if mode == "openai_responses": if not api_key: return StaticTextChatModel(_openai_missing_key_user_message(lang)), mode - return OpenAIResponsesModel(model=model_name, api_key=api_key, base_url=bu or None), mode + return _apply_profile_thinking(OpenAIResponsesModel(model=model_name, api_key=api_key, base_url=bu or None), profile), mode if mode == "anthropic": akey = ( (openai_api_key if allow_runtime_overrides else None) @@ -155,10 +173,10 @@ def _build_executor_components( if mode == "ollama": ollama_base = (bu or DEFAULT_OLLAMA_BASE_URL).strip() or DEFAULT_OLLAMA_BASE_URL ollama_key = api_key or _OLLAMA_DUMMY_KEY - return OpenAIChatModel(model=model_name, api_key=ollama_key, base_url=ollama_base), mode + return _apply_profile_thinking(OpenAIChatModel(model=model_name, api_key=ollama_key, base_url=ollama_base), profile), mode if not api_key: return StaticTextChatModel(_openai_missing_key_user_message(lang)), mode - return OpenAIChatModel(model=model_name, api_key=api_key, base_url=bu or None), mode + return _apply_profile_thinking(OpenAIChatModel(model=model_name, api_key=api_key, base_url=bu or None), profile), mode valid_profile_ids = {p["id"] for p in store.list_llm_profiles(visible_only=True, **list_kw)} if active_pid and active_pid not in valid_profile_ids: diff --git a/runtime/chat/agent_messages.py b/runtime/chat/agent_messages.py index 1397f224..e1469714 100644 --- a/runtime/chat/agent_messages.py +++ b/runtime/chat/agent_messages.py @@ -199,7 +199,28 @@ def build_llm_messages( ) -> list[dict[str, Any]]: """把 DB 中的消息序列转换为 LLM messages。""" out: list[dict[str, Any]] = [{"role": "system", "content": (system_prompt or "").strip()}] + thinking_mode_enabled = bool(getattr(model, "thinking_mode_enabled", False)) allow_signature_replay = _allow_reasoning_signature_replay(model) + reasoning_by_turn: dict[str, list[tuple[int, str]]] = {} + if thinking_mode_enabled: + for m in store_messages or []: + if str(getattr(m, "role", "") or "") != "assistant": + continue + if str(getattr(m, "event_type", "") or "").strip().lower() != "reasoning": + continue + tid = str(getattr(m, "turn_uuid", "") or "").strip() + if not tid: + continue + try: + idx = 0 + ep = getattr(m, "event_payload", None) + if isinstance(ep, str) and ep.strip(): + payload = json.loads(ep) + if isinstance(payload, dict): + idx = int(payload.get("chunk_index") or 0) + except Exception: + idx = 0 + reasoning_by_turn.setdefault(tid, []).append((idx, str(getattr(m, "content", "") or ""))) historical_tool_ids = _collect_historical_tool_call_ids( store_messages=store_messages, full_rounds=_replay_recent_tool_rounds() ) @@ -207,7 +228,46 @@ def build_llm_messages( # that is not present in the assistant tool_calls within the same request context. # This can happen when context windows are trimmed and the assistant tool_calls row is dropped. valid_tool_call_ids: set[str] = set() - for m in store_messages: + # Precompute tool_call_id suffix sets to detect broken tool_calls -> tool pairing. + tool_ids_after: list[set[str]] = [] + seen_tool_ids: set[str] = set() + for m in reversed(store_messages or []): + role = str(getattr(m, "role", "") or "") + if role == "tool": + tcid = _tool_call_id_from_tool_row(getattr(m, "tool_calls", None)) + if tcid: + seen_tool_ids.add(tcid) + tool_ids_after.append(set(seen_tool_ids)) + tool_ids_after.reverse() + def _attach_reasoning_content(row: dict[str, Any], m: Any) -> dict[str, Any]: + if not thinking_mode_enabled: + return row + rc = "" + ep = getattr(m, "event_payload", None) + if isinstance(ep, str) and ep.strip(): + try: + payload = json.loads(ep) + if isinstance(payload, dict): + rc = str(payload.get("reasoning_content") or "").strip() + except Exception: + rc = "" + if not rc: + tid = str(getattr(m, "turn_uuid", "") or "").strip() + chunks = reasoning_by_turn.get(tid) or [] + if chunks: + rc = "\n".join( + [ + str(x[1] or "").strip() + for x in sorted(chunks, key=lambda t: int(t[0] or 0)) + if str(x[1] or "").strip() + ] + ).strip() + # Provider contract: in thinking mode, the field must be present for every assistant message, + # even if empty (some gateways error on missing key). + row["reasoning_content"] = rc + return row + + for i, m in enumerate(store_messages): role = str(getattr(m, "role", "") or "") event_type = str(getattr(m, "event_type", "") or "").strip().lower() if event_type == "reasoning": @@ -408,6 +468,19 @@ def build_llm_messages( tool_calls = None if tool_calls and isinstance(tool_calls, list): + # Guard: only include tool_calls if tool results exist later in this trimmed window. + want_ids = [str(tc.get("id") or "").strip() for tc in tool_calls if isinstance(tc, dict) and str(tc.get("id") or "").strip()] + suffix = tool_ids_after[i] if (i >= 0 and i < len(tool_ids_after)) else set() + if want_ids and any(tid not in suffix for tid in want_ids): + tool_calls = None + if tool_calls is None: + out.append( + _attach_reasoning_content( + {"role": "assistant", "content": _strip_reasoning_blocks(getattr(m, "content", "") or "")}, + m, + ) + ) + continue api_tool_calls = [] gemini_fc = gemini_openai_compat_client(model) for idx, tc in enumerate(tool_calls): @@ -439,16 +512,29 @@ def build_llm_messages( api_tool_calls.append(entry) if api_tool_calls: out.append( - { - "role": "assistant", - "content": _strip_reasoning_blocks(getattr(m, "content", "") or ""), - "tool_calls": api_tool_calls, - } + _attach_reasoning_content( + { + "role": "assistant", + "content": _strip_reasoning_blocks(getattr(m, "content", "") or ""), + "tool_calls": api_tool_calls, + }, + m, + ) ) else: - out.append({"role": "assistant", "content": _strip_reasoning_blocks(getattr(m, "content", "") or "")}) + out.append( + _attach_reasoning_content( + {"role": "assistant", "content": _strip_reasoning_blocks(getattr(m, "content", "") or "")}, + m, + ) + ) else: - out.append({"role": "assistant", "content": _strip_reasoning_blocks(getattr(m, "content", "") or "")}) + out.append( + _attach_reasoning_content( + {"role": "assistant", "content": _strip_reasoning_blocks(getattr(m, "content", "") or "")}, + m, + ) + ) continue if role == "tool": diff --git a/runtime/direct_loop.py b/runtime/direct_loop.py index a22706b8..4a96d67d 100644 --- a/runtime/direct_loop.py +++ b/runtime/direct_loop.py @@ -807,6 +807,7 @@ def _persist_assistant_step( assistant_text: str, reasoning_text: str, llm_tool_calls: list[Any], + thinking_mode_enabled: bool = False, ) -> _LoopStepResult: stored_tool_calls = [] for tc in llm_tool_calls: @@ -819,19 +820,18 @@ def _persist_assistant_step( } ) - reasoning_chunks, assistant_body = _split_reasoning_and_body( - assistant_text, - explicit_reasoning=reasoning_text, - ) - for idx, chunk in enumerate(reasoning_chunks): - store.add_message( - session_id=session_id, - role="assistant", - content=chunk, - turn_uuid=turn_uuid, - event_type="reasoning", - event_payload={"chunk_index": int(idx), "chunk_count": len(reasoning_chunks)}, - ) + reasoning_chunks, assistant_body = _split_reasoning_and_body(assistant_text, explicit_reasoning=reasoning_text) + reasoning_full = "\n".join([str(x or "").strip() for x in reasoning_chunks if str(x or "").strip()]).strip() + if not thinking_mode_enabled: + for idx, chunk in enumerate(reasoning_chunks): + store.add_message( + session_id=session_id, + role="assistant", + content=chunk, + turn_uuid=turn_uuid, + event_type="reasoning", + event_payload={"chunk_index": int(idx), "chunk_count": len(reasoning_chunks)}, + ) assistant_row = store.add_message( session_id=session_id, role="assistant", @@ -839,6 +839,7 @@ def _persist_assistant_step( tool_calls=stored_tool_calls or None, turn_uuid=turn_uuid, event_type="tool_call" if stored_tool_calls else "assistant_text", + event_payload=({"reasoning_content": reasoning_full} if (thinking_mode_enabled and reasoning_full) else None), ) return _LoopStepResult( assistant_text=assistant_body, @@ -1002,6 +1003,7 @@ def run_oclaw_direct_loop( assistant_text=assistant_text, reasoning_text=reasoning_text, llm_tool_calls=llm_tool_calls, + thinking_mode_enabled=bool(getattr(model, "thinking_mode_enabled", False)), ) final_text = step.assistant_text if not step.llm_tool_calls: @@ -1082,6 +1084,7 @@ def run_oclaw_direct_loop( assistant_text=str(getattr(resp, "content", "") or ""), reasoning_text=str(getattr(resp, "reasoning_content", "") or ""), llm_tool_calls=[], + thinking_mode_enabled=bool(getattr(model, "thinking_mode_enabled", False)), ) final_text = step.assistant_text diff --git a/runtime/tools/mcp/runtime.py b/runtime/tools/mcp/runtime.py index 1f66931b..5740f6f0 100644 --- a/runtime/tools/mcp/runtime.py +++ b/runtime/tools/mcp/runtime.py @@ -196,14 +196,57 @@ class McpProcessRuntime: req = {"jsonrpc": "2.0", "id": rid, "method": str(method), "params": params or {}} p.stdin.write(json.dumps(req, ensure_ascii=False) + "\n") p.stdin.flush() + skipped: list[str] = [] + max_skip = 60 while True: line = p.stdout.readline() if not line: - return {"ok": False, "error_code": "mcp_runtime_empty_response", "error": "empty_response"} + # Process may have exited early (common when runtime deps are missing). + rc = None + try: + rc = p.poll() + except Exception: + rc = None + err_tail = "" + if rc is not None and p.stderr is not None: + try: + err_tail = (p.stderr.read() or "")[-2000:] + except Exception: + err_tail = "" + out: dict[str, Any] = {"ok": False, "error_code": "mcp_runtime_empty_response", "error": "empty_response"} + if rc is not None: + out["exit_code"] = int(rc) + if err_tail.strip(): + out["stderr_tail"] = err_tail.strip() + return out + s = str(line).strip() + if not s: + # Some servers emit blank lines; ignore. + continue + # Some MCP servers print logs/banner to stdout. Skip non-JSON lines until a JSON-RPC response arrives. + if not (s.startswith("{") or s.startswith("[")): + skipped.append(s[:200]) + if len(skipped) > max_skip: + return { + "ok": False, + "error_code": "mcp_runtime_protocol_mismatch", + "error": "non_jsonrpc_response", + "skipped": skipped[-12:], + } + continue try: - obj = json.loads(line) + obj = json.loads(s) except Exception as exc: - return {"ok": False, "error_code": "mcp_runtime_bad_json", "error": str(exc)} + # If a server mixes JSON with log fragments, keep skipping until we see a clean JSON object. + skipped.append(s[:200]) + if len(skipped) > max_skip: + return { + "ok": False, + "error_code": "mcp_runtime_bad_json", + "error": str(exc), + "skipped": skipped[-12:], + } + continue if not isinstance(obj, dict): return {"ok": False, "error_code": "mcp_runtime_invalid_payload", "error": "response_not_object"} if "jsonrpc" not in obj and "id" not in obj: diff --git a/runtime/workspaces/generalist/ROLE_SYSTEM.md b/runtime/workspaces/generalist/ROLE_SYSTEM.md index 852c9e7e..c6e8e9e2 100644 --- a/runtime/workspaces/generalist/ROLE_SYSTEM.md +++ b/runtime/workspaces/generalist/ROLE_SYSTEM.md @@ -16,4 +16,5 @@ ## 主要事项: - 工具失败时先报告 `error_code` 与原因,再给下一步。 +- Windows(PowerShell/CMD)如遇部分命令“空输出/乱码”,优先尝试 `chcp 65001 > nul && `。 - 禁止伪造工具结果。