mirror of
https://github.com/hansjone/oclaw.git
synced 2026-10-09 01:50:44 +08:00
Fix WhatsApp same-session quote re-injection.
Strip @mention prefixes and widen history lookup so reply-to of the bot's own prior answer is not wrapped as [被引用消息] again. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
parent
2446993b11
commit
10ae4abc1d
3 changed files with 153 additions and 19 deletions
|
|
@ -1318,7 +1318,9 @@ def process_inbound_payload(payload: dict[str, Any]) -> dict[str, Any]:
|
|||
quoted_info = extract_group_quoted_message(metadata=meta_for_group)
|
||||
quoted_text = str(quoted_info.get("quoted_text") or "").strip()
|
||||
if quoted_text:
|
||||
recent_messages = store.get_messages(str(session_id), limit=6)
|
||||
# Look back far enough to cover long tool loops; dedupe uses
|
||||
# assistant bodies so same-session reply-to will not re-inject.
|
||||
recent_messages = store.get_messages(str(session_id), limit=80)
|
||||
if should_inject_quoted_context(
|
||||
quoted_text=quoted_text,
|
||||
recent_messages=recent_messages,
|
||||
|
|
|
|||
|
|
@ -169,19 +169,75 @@ def extract_group_quoted_message(*, metadata: dict[str, Any] | None) -> dict[str
|
|||
|
||||
|
||||
def _normalize_quoted_compare_text(text: str) -> str:
|
||||
s = re.sub(r"\s+", " ", str(text or "").strip())
|
||||
return s[:280]
|
||||
"""Normalize quote/history text for dedupe.
|
||||
|
||||
WhatsApp reply-to often prefixes the bot body with ``@<lid|phone>``; session
|
||||
assistant rows usually do not. Strip those so same-session quotes match.
|
||||
"""
|
||||
s = str(text or "").strip()
|
||||
if not s:
|
||||
return ""
|
||||
# Drop leading @tokens (jid / phone / display-name mentions).
|
||||
s = re.sub(r"^(?:@\S+\s+)+", "", s)
|
||||
# Drop common group-ingest wrappers if somehow present.
|
||||
s = re.sub(r"^\[被引用消息\]\s*", "", s)
|
||||
s = re.sub(r"\s+", " ", s).strip()
|
||||
return s[:400]
|
||||
|
||||
|
||||
def _quoted_texts_overlap(quoted: str, history: str) -> bool:
|
||||
"""True when quote body is essentially the same as a history row."""
|
||||
q = _normalize_quoted_compare_text(quoted)
|
||||
h = _normalize_quoted_compare_text(history)
|
||||
if not q or not h:
|
||||
return False
|
||||
if q == h:
|
||||
return True
|
||||
q_fp = q[:160]
|
||||
h_fp = h[:160]
|
||||
if q_fp == h_fp:
|
||||
return True
|
||||
# Substantial containment only — avoid short-string false positives.
|
||||
min_len = 48
|
||||
if len(q_fp) >= min_len and q_fp in h:
|
||||
return True
|
||||
if len(h_fp) >= min_len and h_fp in q:
|
||||
return True
|
||||
if len(q) >= min_len and len(h) >= min_len:
|
||||
n = 0
|
||||
for a, b in zip(q_fp, h_fp):
|
||||
if a != b:
|
||||
break
|
||||
n += 1
|
||||
if n >= min_len:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def should_inject_quoted_context(*, quoted_text: str, recent_messages: list[Any]) -> bool:
|
||||
"""Return False when quoted body already appears in recent session history.
|
||||
|
||||
Prefer assistant rows (same-session bot replies). Also scan other roles as a
|
||||
fallback so previously injected ``[被引用消息]`` user rows still dedupe.
|
||||
"""
|
||||
token = _normalize_quoted_compare_text(quoted_text)
|
||||
if not token:
|
||||
return False
|
||||
|
||||
assistants: list[Any] = []
|
||||
others: list[Any] = []
|
||||
for row in recent_messages or []:
|
||||
content = _normalize_quoted_compare_text(getattr(row, "content", "") or "")
|
||||
if not content:
|
||||
continue
|
||||
if token == content or token in content or content in token:
|
||||
role = str(getattr(row, "role", "") or "").strip().lower()
|
||||
if role == "assistant":
|
||||
assistants.append(row)
|
||||
else:
|
||||
others.append(row)
|
||||
|
||||
for row in assistants:
|
||||
if _quoted_texts_overlap(token, str(getattr(row, "content", "") or "")):
|
||||
return False
|
||||
for row in others:
|
||||
if _quoted_texts_overlap(token, str(getattr(row, "content", "") or "")):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
|
|
|||
|
|
@ -443,13 +443,73 @@ def test_should_inject_quoted_context_dedupes_recent_message() -> None:
|
|||
id=1,
|
||||
session_id="s1",
|
||||
role="assistant",
|
||||
content="服务器刚刚 502 了",
|
||||
content="server just returned HTTP 502",
|
||||
tool_calls=None,
|
||||
timestamp="",
|
||||
)
|
||||
]
|
||||
assert should_inject_quoted_context(quoted_text="服务器刚刚 502 了", recent_messages=recent) is False
|
||||
assert should_inject_quoted_context(quoted_text="另一条内容", recent_messages=recent) is True
|
||||
assert should_inject_quoted_context(quoted_text="server just returned HTTP 502", recent_messages=recent) is False
|
||||
assert should_inject_quoted_context(quoted_text="a totally different question", recent_messages=recent) is True
|
||||
|
||||
|
||||
def test_should_inject_quoted_context_strips_whatsapp_mention_prefix() -> None:
|
||||
from svc.persistence.sqlite_store import ChatMessage
|
||||
|
||||
body = (
|
||||
"### KND-SMNR-EN1-Z20HS — LLDP Neighbors\n\n"
|
||||
"| Device | Port | Peer Device |\n"
|
||||
"|---|---|---|\n"
|
||||
"| KND-SMNR-EN1-Z20HS | cgei-1/1/0/33 | KND-LGGO-EN1-Z20HS |\n"
|
||||
"KND-SMNR-EN1 has only 2 LLDP neighbors in the Kendahe area.\n"
|
||||
)
|
||||
recent = [
|
||||
ChatMessage(
|
||||
id=10,
|
||||
session_id="s1",
|
||||
role="assistant",
|
||||
content=body,
|
||||
tool_calls=None,
|
||||
timestamp="",
|
||||
event_type="assistant_text",
|
||||
)
|
||||
]
|
||||
# WhatsApp reply-to often prefixes the bot body with @<bot lid/jid>.
|
||||
quoted = f"@68458037407987 {body}"
|
||||
assert should_inject_quoted_context(quoted_text=quoted, recent_messages=recent) is False
|
||||
|
||||
|
||||
def test_should_inject_quoted_context_finds_assistant_buried_under_tools() -> None:
|
||||
from svc.persistence.sqlite_store import ChatMessage
|
||||
|
||||
body = (
|
||||
"KND-SMNR-EN1 has only 2 LLDP neighbors — both access-layer devices "
|
||||
"in the Kendahe area connected via cgei ports."
|
||||
)
|
||||
rows = [
|
||||
ChatMessage(
|
||||
id=1,
|
||||
session_id="s1",
|
||||
role="assistant",
|
||||
content=body,
|
||||
tool_calls=None,
|
||||
timestamp="",
|
||||
event_type="assistant_text",
|
||||
)
|
||||
]
|
||||
for i in range(2, 20):
|
||||
rows.append(
|
||||
ChatMessage(
|
||||
id=i,
|
||||
session_id="s1",
|
||||
role="tool",
|
||||
content=f'{{"ok": true, "n": {i}}}',
|
||||
tool_calls=None,
|
||||
timestamp="",
|
||||
event_type="tool_result",
|
||||
)
|
||||
)
|
||||
assert should_inject_quoted_context(quoted_text=f"@bot {body}", recent_messages=rows) is False
|
||||
assert should_inject_quoted_context(quoted_text="totally new topic xyz about fans", recent_messages=rows) is True
|
||||
|
||||
|
||||
def test_build_whatsapp_group_reply_metadata() -> None:
|
||||
|
|
@ -586,6 +646,8 @@ def test_inbound_dm_still_processes_without_mention(monkeypatch: pytest.MonkeyPa
|
|||
store = fresh_sqlite_store
|
||||
_setup_whatsapp_identity(store)
|
||||
monkeypatch.setattr("svc.persistence.assistant_store.get_assistant_store", lambda: store)
|
||||
# Keep sync replies so this unit test can assert HTTP response shape.
|
||||
monkeypatch.setenv("OCLAW_WHATSAPP_INBOUND_QUEUE_DELIVERY", "0")
|
||||
|
||||
captured: dict[str, str] = {}
|
||||
|
||||
|
|
@ -835,23 +897,35 @@ def test_inbound_group_skips_quoted_context_when_already_in_current_session(
|
|||
|
||||
call_no = {"n": 0}
|
||||
captured: dict[str, str] = {}
|
||||
prior_reply = "Looks like OSPF neighbor flapping"
|
||||
|
||||
class _Turn:
|
||||
turn_uuid = "turn-q2"
|
||||
|
||||
@property
|
||||
def reply_text(self) -> str:
|
||||
return "看起来像 OSPF 邻居抖动" if call_no["n"] == 1 else "继续分析"
|
||||
return prior_reply if call_no["n"] == 1 else "continue analysis"
|
||||
|
||||
class _Gw:
|
||||
def __init__(self, *, store: object) -> None:
|
||||
_ = store
|
||||
self._store = store
|
||||
|
||||
def handle_turn(self, **kwargs: object) -> _Turn:
|
||||
call_no["n"] += 1
|
||||
msg = kwargs.get("msg")
|
||||
captured["text"] = str(getattr(msg, "text", "") or "")
|
||||
return _Turn()
|
||||
turn = _Turn()
|
||||
# Real gateway persists assistant text; mock must do the same so quote
|
||||
# dedupe can see the prior reply in session history.
|
||||
sid = str(getattr(msg, "session_id", "") or "").strip()
|
||||
if sid and turn.reply_text:
|
||||
self._store.add_message(
|
||||
sid,
|
||||
"assistant",
|
||||
turn.reply_text,
|
||||
event_type="assistant_text",
|
||||
)
|
||||
return turn
|
||||
|
||||
monkeypatch.setattr("runtime.gateway.OclawGateway", _Gw)
|
||||
|
||||
|
|
@ -861,7 +935,7 @@ def test_inbound_group_skips_quoted_context_when_already_in_current_session(
|
|||
"account_id": "wa-default",
|
||||
"user_id": "111@s.whatsapp.net",
|
||||
"chat_id": "120363012345678@g.us",
|
||||
"text": "@bot 先帮我判断原因",
|
||||
"text": "@bot please diagnose first",
|
||||
"is_group": True,
|
||||
"mentions": ["999@s.whatsapp.net"],
|
||||
"metadata": {"bot_jid": "999@s.whatsapp.net", "raw": {"pushName": "Alice"}},
|
||||
|
|
@ -873,14 +947,15 @@ def test_inbound_group_skips_quoted_context_when_already_in_current_session(
|
|||
"account_id": "wa-default",
|
||||
"user_id": "111@s.whatsapp.net",
|
||||
"chat_id": "120363012345678@g.us",
|
||||
"text": "@bot 那下一步怎么排查?",
|
||||
"text": "@bot what is the next check?",
|
||||
"is_group": True,
|
||||
"mentions": ["999@s.whatsapp.net"],
|
||||
"metadata": {
|
||||
"bot_jid": "999@s.whatsapp.net",
|
||||
"raw": {
|
||||
"pushName": "Alice",
|
||||
"quotedText": "看起来像 OSPF 邻居抖动",
|
||||
# WhatsApp reply-to often prefixes the bot body with @<bot jid/lid>.
|
||||
"quotedText": f"@999 {prior_reply}",
|
||||
"quotedParticipant": "999@s.whatsapp.net",
|
||||
"quotedPushName": "oclaw",
|
||||
"quotedStanzaId": "Q3",
|
||||
|
|
@ -899,10 +974,11 @@ def test_inbound_group_reply_includes_quote_and_mention_metadata(
|
|||
store = fresh_sqlite_store
|
||||
_setup_whatsapp_identity(store)
|
||||
monkeypatch.setattr("svc.persistence.assistant_store.get_assistant_store", lambda: store)
|
||||
monkeypatch.setenv("OCLAW_WHATSAPP_INBOUND_QUEUE_DELIVERY", "0")
|
||||
|
||||
class _Turn:
|
||||
turn_uuid = "turn-g"
|
||||
reply_text = "下午三点"
|
||||
reply_text = "afternoon three"
|
||||
|
||||
class _Gw:
|
||||
def __init__(self, *, store: object) -> None:
|
||||
|
|
@ -919,7 +995,7 @@ def test_inbound_group_reply_includes_quote_and_mention_metadata(
|
|||
"account_id": "wa-default",
|
||||
"user_id": "111@s.whatsapp.net",
|
||||
"chat_id": "120363012345678@g.us",
|
||||
"text": "@bot 明天几点?",
|
||||
"text": "@bot tomorrow when?",
|
||||
"is_group": True,
|
||||
"mentions": ["999@s.whatsapp.net"],
|
||||
"metadata": {
|
||||
|
|
@ -934,7 +1010,7 @@ def test_inbound_group_reply_includes_quote_and_mention_metadata(
|
|||
assert isinstance(meta, dict)
|
||||
assert meta.get("quote_stanza_id") == "ABC123"
|
||||
assert meta.get("mention_jids") == ["111@s.whatsapp.net"]
|
||||
assert meta.get("quote_text") == "@bot 明天几点?"
|
||||
assert meta.get("quote_text") == "@bot tomorrow when?"
|
||||
|
||||
|
||||
def test_inbound_group_schedule_mention_metadata_preserved_after_text_clean(
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue