mirror of
https://github.com/hansjone/dsh-im-ops.git
synced 2026-10-11 01:43:25 +08:00
fix(feishu): renew the reply wait window while the turn is active
A long-running Harness task was falsely reported as MODEL_REPLY_TIMEOUT after the fixed 10-minute deadline even though the backend was still working. Replace the hard wall-clock deadline with an activity-based stall window: the wait renews whenever the turn produces new events, the control ownership is active, or (mid-turn with no new events) Harness still reports the session as running. Only a genuine stall with no progress for the whole timeoutMs window fails fast. This is the shared harness-client layer, exercised end-to-end by the Feishu channel (which already exposes HARNESS_REPLY_TIMEOUT_MS); other channels inherit the same fix. Adds two regression tests: a long-running turn with periodic activity completes instead of timing out, and a genuinely stalled turn still fails with harness-reply-timeout.
This commit is contained in:
parent
c2be2389b0
commit
955c2217fd
3 changed files with 132 additions and 18 deletions
|
|
@ -428,6 +428,11 @@ export class HarnessReplyTracker {
|
|||
return this.#finished;
|
||||
}
|
||||
|
||||
/** The highest event seq consumed so far; advances as the turn produces events. */
|
||||
get lastSeq() {
|
||||
return this.#lastSeq;
|
||||
}
|
||||
|
||||
get answer() {
|
||||
return this.#latestText.trim();
|
||||
}
|
||||
|
|
@ -1387,8 +1392,13 @@ export class HarnessClient {
|
|||
promptAccepted = true;
|
||||
|
||||
try {
|
||||
const deadline = Date.now() + timeoutMs;
|
||||
while (Date.now() < deadline) {
|
||||
// Renew the wait window whenever the turn is still making progress, so
|
||||
// a long-running task is not falsely reported as MODEL_REPLY_TIMEOUT
|
||||
// while Harness is still working. A genuine stall (no new events and
|
||||
// no active turn for the whole timeoutMs window) still fails fast.
|
||||
let lastProgressAt = Date.now();
|
||||
let lastPollSeq = tracker.lastSeq;
|
||||
while (Date.now() - lastProgressAt < timeoutMs) {
|
||||
await sleep(300, signal);
|
||||
const history = await this.rpc(
|
||||
'session.history',
|
||||
|
|
@ -1402,6 +1412,26 @@ export class HarnessClient {
|
|||
if (!wasActive && ownership.active) ownership.reconnect?.();
|
||||
}
|
||||
const updates = tracker.consumeAll(history.events ?? []);
|
||||
// Progress = the turn produced new events, or the ownership is still
|
||||
// active. Either refreshes the stall window.
|
||||
const seqAdvanced = tracker.lastSeq > lastPollSeq;
|
||||
lastPollSeq = tracker.lastSeq;
|
||||
if (seqAdvanced) {
|
||||
lastProgressAt = Date.now();
|
||||
} else if (ownership?.active) {
|
||||
lastProgressAt = Date.now();
|
||||
} else if (tracker.tracking) {
|
||||
// Mid-turn but no new events: ask Harness whether the session is
|
||||
// still running before declaring a stall. A failing probe must not
|
||||
// mask a real stall, so it is not treated as progress.
|
||||
try {
|
||||
if (await this.isSessionRunning(sessionId, { signal })) {
|
||||
lastProgressAt = Date.now();
|
||||
}
|
||||
} catch {
|
||||
// ignore probe failure; the deadline still applies
|
||||
}
|
||||
}
|
||||
if (onUpdate) {
|
||||
const visibleUpdates = progressMode === 'all' ? updates : updates.slice(-1);
|
||||
for (const update of visibleUpdates) {
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue