diff --git a/extensions/pi/mempalace.ts b/extensions/pi/mempalace.ts index 1371116..58d699d 100644 --- a/extensions/pi/mempalace.ts +++ b/extensions/pi/mempalace.ts @@ -855,7 +855,22 @@ export default async function mempalaceExtension(pi: ExtensionAPI) { const feedWing = process.env.MEMPALACE_FEED_WING ?? "wing_conversations"; const feedDebounceMs = num(process.env.MEMPALACE_FEED_DEBOUNCE_MS, 600_000); const feedPrepareTimeoutMs = num(process.env.MEMPALACE_FEED_PREPARE_TIMEOUT_MS, 120_000); - const feedMineTimeoutMs = num(process.env.MEMPALACE_FEED_MINE_TIMEOUT_MS, 30_000); + // 300_000, raised from 30_000 on 2026-09-10. The mine is the SLOWEST thing + // this extension does — it offers every qualifying session transcript to a + // SINGLE-WRITER palace, measured at 30–60s in normal operation and growing + // with the corpus — yet it carried by far the TIGHTEST deadline: 4x tighter + // than the prepare step that precedes it (120_000) and 10x tighter than the + // init handshake (300_000), which is a fast call. All three were introduced + // together in 29e660e (2026-08-12) and this one was never revisited, so the + // deadline fired during entirely normal operation and the resulting message + // read as an error when nothing had gone wrong. + // + // Matched to the init timeout because liveness is the ONLY legitimate job + // left for this deadline: the Promise.race below abandons our WAIT, it cannot + // cancel the server's work, so the deadline buys nothing except an escape + // from a permanently hung call. It must therefore sit far above the slowest + // honest completion, not near it. + const feedMineTimeoutMs = num(process.env.MEMPALACE_FEED_MINE_TIMEOUT_MS, 300_000); let lastFeedAt = 0; // 0 => the first settled turn also acts as a catch-up let feedInFlight: Promise | null = null; @@ -909,6 +924,23 @@ export default async function mempalaceExtension(pi: ExtensionAPI) { try { const source = await prepareFeed(reason); if (!source) return; + // Record the attempt HERE, before awaiting — not after a successful + // wait. The race below abandons only our WAIT; the mine keeps running + // server-side, and `mine --mode convos` dedups by source_file and is + // idempotent, so a timeout is emphatically not a "did not happen". + // + // Leaving lastFeedAt stale on the timeout path defeated the debounce + // guard in the agent_settled handler below (`Date.now() - lastFeedAt < + // feedDebounceMs`): with lastFeedAt unchanged that guard passed on EVERY + // settled turn, and because `run` had already settled, feedInFlight was + // null too — so BOTH guards stood open. Each settled turn then launched + // another mine while the previous one was still running: overlapping + // writers queueing on a single-writer palace, each making the next one + // slower and the next timeout likelier. That positive feedback loop, + // not the tight deadline by itself, is why operators saw "mine timed out + // after 30000ms" many times per session rather than at most once per + // debounce window. + lastFeedAt = Date.now(); await Promise.race([ client.callTool("mempalace_mine", { source, @@ -927,7 +959,6 @@ export default async function mempalaceExtension(pi: ExtensionAPI) { ), ), ]); - lastFeedAt = Date.now(); } catch (err) { process.stderr.write( `[mempalace ext] feed (${reason}) failed: ${(err as Error).message}\n`,