fix(feed): the mine's deadline never reached the transport; 60 s cut it off
The 2026-09 change raised MEMPALACE_FEED_MINE_TIMEOUT_MS to 300 000 and
raced it against client.callTool("mempalace_mine"). But callTool() had no
way to carry a deadline, so every call went out under the transport's
generic per-request timeout (MEMPALACE_MCP_TIMEOUT_MS, 60 000), which fired
first on every honest 60 s+ mine. Operators saw
feed (tick) failed: mempalace remote request 'tools/call' failed:
timed out after 60000ms
instead of the message the change had aimed at, and the 300 s was
unreachable. On stdio it was worse than noise: that transport kills the
server child on timeout, so the mine was actually aborted at 60 s.
callTool(name, args, { timeoutMs }) now passes a per-call deadline to both
transports; feedPalace() uses it for the mine. Plain calls keep the short
default — a query taking 60 s is still wedged. The Promise.race stays as the
liveness guard for a transport with its timeout disabled (0).
scripts/test-mcp-call-timeout.sh cuts RemoteMcpClient out of the shipped
file (as test-owed-withdrawal.sh does for the mailbox predicates), drives it
against a local JSON-RPC server that delays tools/call, and asserts: plain
call rejects at the generic deadline; the override outlives it; the override
is itself a deadline. Fails on the previous commit (2 of 6), passes here.
Vendored-copy delta noted in the header; protocol untouched, sync token
unchanged (check-mcp-client-sync.sh passes against pi-extensions).
This commit is contained in:
+44
-16
@@ -67,7 +67,9 @@
|
||||
* child, so pi gets an error instead of hanging and later calls fail fast.
|
||||
* This is a per-REQUEST timeout, not a process-lifetime one — the
|
||||
* long-lived server is only killed when a request genuinely stalls.
|
||||
* - MEMPALACE_MCP_TIMEOUT_MS tool-call/request timeout (default 60000)
|
||||
* - MEMPALACE_MCP_TIMEOUT_MS tool-call/request timeout (default 60000);
|
||||
* the feed's mine carries its own, longer
|
||||
* deadline (MEMPALACE_FEED_MINE_TIMEOUT_MS)
|
||||
* - MEMPALACE_MCP_INIT_TIMEOUT_MS initialize+tools/list timeout (default 300000)
|
||||
* Set either to 0 to disable (legacy unbounded behavior).
|
||||
*
|
||||
@@ -116,7 +118,13 @@ interface IMcpClient {
|
||||
readonly alive: boolean;
|
||||
onExit: (() => void) | null;
|
||||
start(): Promise<void>;
|
||||
callTool(name: string, args: Record<string, unknown>): Promise<any>;
|
||||
/**
|
||||
* `opts.timeoutMs` overrides the transport's generic per-request deadline
|
||||
* for THIS call only. Callers that knowingly invoke a long server-side job
|
||||
* (the feed's `mempalace_mine`) pass their own deadline here; everything
|
||||
* else keeps the short default, which is the wedged-query guard.
|
||||
*/
|
||||
callTool(name: string, args: Record<string, unknown>, opts?: { timeoutMs?: number }): Promise<any>;
|
||||
ensureAlive(): Promise<boolean>;
|
||||
stop(): void | Promise<void>;
|
||||
}
|
||||
@@ -372,8 +380,12 @@ class StdioMcpClient implements IMcpClient {
|
||||
});
|
||||
}
|
||||
|
||||
async callTool(name: string, args: Record<string, unknown>): Promise<any> {
|
||||
return this.request("tools/call", { name, arguments: args });
|
||||
async callTool(name: string, args: Record<string, unknown>, opts?: { timeoutMs?: number }): Promise<any> {
|
||||
// The per-call override matters MORE here than for the HTTP client: on
|
||||
// timeout this transport kills the server child, so a generic deadline
|
||||
// that undercuts a long mine does not merely abandon the wait — it
|
||||
// aborts the mine.
|
||||
return this.request("tools/call", { name, arguments: args }, opts?.timeoutMs ?? this.requestTimeoutMs);
|
||||
}
|
||||
|
||||
/** SIGTERM then SIGKILL grace, for stall recovery. */
|
||||
@@ -421,6 +433,10 @@ class StdioMcpClient implements IMcpClient {
|
||||
// • per-request AbortController timeout honouring MEMPALACE_MCP_TIMEOUT_MS /
|
||||
// MEMPALACE_MCP_INIT_TIMEOUT_MS, mirroring StdioMcpClient's timeout ethos.
|
||||
// • alive / ensureAlive / onExit to satisfy IMcpClient.
|
||||
// • callTool() takes an optional per-call `{ timeoutMs }` (IMcpClient
|
||||
// contract, see there) so the feed's long-running mine is not cut off by
|
||||
// the generic per-request deadline. Not a protocol change; sync token
|
||||
// unchanged.
|
||||
//
|
||||
// NOTE: mempalace-mcp --transport http is a SESSIONLESS, stateless JSON-RPC
|
||||
// server (no Mcp-Session-Id, always application/json, Connection: close), so
|
||||
@@ -466,8 +482,8 @@ class RemoteMcpClient implements IMcpClient {
|
||||
this.healthy = true;
|
||||
}
|
||||
|
||||
async callTool(name: string, args: Record<string, unknown>): Promise<any> {
|
||||
return this.request("tools/call", { name, arguments: args });
|
||||
async callTool(name: string, args: Record<string, unknown>, opts?: { timeoutMs?: number }): Promise<any> {
|
||||
return this.request("tools/call", { name, arguments: args }, { timeoutMs: opts?.timeoutMs });
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -941,17 +957,29 @@ export default async function mempalaceExtension(pi: ExtensionAPI) {
|
||||
// after 30000ms" many times per session rather than at most once per
|
||||
// debounce window.
|
||||
lastFeedAt = Date.now();
|
||||
// The deadline is passed DOWN to the transport as well as raced
|
||||
// here. Until 2026-09-18 it was only raced: callTool() had no way
|
||||
// to carry it, so the transport's generic per-request timeout
|
||||
// (MEMPALACE_MCP_TIMEOUT_MS, 60 000) fired first on every honest
|
||||
// 60 s+ mine — "remote request 'tools/call' failed: timed out
|
||||
// after 60000ms" — and the 300 000 below was unreachable. The
|
||||
// race stays as the liveness guard for a transport whose timeout
|
||||
// is disabled (0).
|
||||
await Promise.race([
|
||||
client.callTool("mempalace_mine", {
|
||||
source,
|
||||
mode: "convos",
|
||||
wing: feedWing,
|
||||
// Internal call: it does not pass through the registered tool's
|
||||
// execute(), so it stamps itself. These ARE this harness's own
|
||||
// transcripts from this device, so the harness segment is the
|
||||
// agent (not `miner`) even though the tool is `mine`.
|
||||
agent: stampProvenance ? `${agentName}@${device}` : agentName,
|
||||
}),
|
||||
client.callTool(
|
||||
"mempalace_mine",
|
||||
{
|
||||
source,
|
||||
mode: "convos",
|
||||
wing: feedWing,
|
||||
// Internal call: it does not pass through the registered tool's
|
||||
// execute(), so it stamps itself. These ARE this harness's own
|
||||
// transcripts from this device, so the harness segment is the
|
||||
// agent (not `miner`) even though the tool is `mine`.
|
||||
agent: stampProvenance ? `${agentName}@${device}` : agentName,
|
||||
},
|
||||
{ timeoutMs: feedMineTimeoutMs },
|
||||
),
|
||||
new Promise((_resolve, reject) =>
|
||||
setTimeout(
|
||||
() => reject(new Error(`mine timed out after ${feedMineTimeoutMs}ms`)),
|
||||
|
||||
Reference in New Issue
Block a user