mirror of
https://github.com/rowboatlabs/rowboat.git
synced 2026-07-24 21:41:08 +02:00
feat(x): graceful wrap-up when a chat turn hits its model-call limit
Interactive turns no longer die with a raw 'Model call limit of N reached' error. The final budgeted model call is composed as a wrap-up: the durable request records wrapUp: true, and the composer strips tools and appends a notice telling the model to answer with what it has, state what's unfinished, and mention the limit is configurable in Settings -> Advanced. The turn then completes normally with that answer. Headless turns (and a wrap-up call that still emits tool calls) keep the hard model-call-limit failure so automation retains the machine-readable outcome; sub-agents already surface partial results. If a hard limit failure still reaches the chat UI, the error bubble now explains the limit and points at the setting instead of the raw error.
This commit is contained in:
parent
47453fc333
commit
f270f39e35
6 changed files with 129 additions and 2 deletions
|
|
@ -6,6 +6,7 @@ import type {
|
||||||
ToolPermissionRequestEvent,
|
ToolPermissionRequestEvent,
|
||||||
} from '@x/shared/src/runs.js'
|
} from '@x/shared/src/runs.js'
|
||||||
import {
|
import {
|
||||||
|
MODEL_CALL_LIMIT_ERROR_CODE,
|
||||||
deriveTurnStatus,
|
deriveTurnStatus,
|
||||||
outstandingAsyncTools,
|
outstandingAsyncTools,
|
||||||
outstandingPermissions,
|
outstandingPermissions,
|
||||||
|
|
@ -331,10 +332,17 @@ export function buildTurnConversation(state: TurnState): ConversationItem[] {
|
||||||
}
|
}
|
||||||
|
|
||||||
if (state.terminal?.type === 'turn_failed') {
|
if (state.terminal?.type === 'turn_failed') {
|
||||||
|
// Interactive turns normally wrap up gracefully before hitting the
|
||||||
|
// limit; if a hard limit failure still lands here, explain it and point
|
||||||
|
// at the setting instead of showing the raw runtime error.
|
||||||
|
const message = state.terminal.code === MODEL_CALL_LIMIT_ERROR_CODE
|
||||||
|
? `This turn stopped after reaching its model-call limit of ${state.definition.config.maxModelCalls}. ` +
|
||||||
|
'Work completed so far is saved above. You can raise the limit in Settings → Advanced.'
|
||||||
|
: state.terminal.error
|
||||||
items.push({
|
items.push({
|
||||||
id: `${turnId}:error`,
|
id: `${turnId}:error`,
|
||||||
kind: 'error',
|
kind: 'error',
|
||||||
message: state.terminal.error,
|
message,
|
||||||
timestamp: ts(),
|
timestamp: ts(),
|
||||||
} satisfies ErrorMessage)
|
} satisfies ErrorMessage)
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -519,6 +519,13 @@ Rules:
|
||||||
the global limit. Without a resolver (tests) it defaults to `20`. Spawned
|
the global limit. Without a resolver (tests) it defaults to `20`. Spawned
|
||||||
sub-agents default to the global limit, which also caps the budget a
|
sub-agents default to the global limit, which also caps the budget a
|
||||||
parent may grant them. Settings changes affect only newly created turns.
|
parent may grant them. Settings changes affect only newly created turns.
|
||||||
|
- Budget exhaustion (§20): interactive turns (`humanAvailable: true`) spend
|
||||||
|
their final budgeted call wrapping up — the request records `wrapUp: true`,
|
||||||
|
and the composer strips tools and appends the shared `wrapUpNotice`, so
|
||||||
|
the turn completes with a real answer that admits what's unfinished.
|
||||||
|
Headless turns (and a wrap-up call that still emits tool calls) keep the
|
||||||
|
hard `turn_failed` with `model-call-limit` — automation needs the
|
||||||
|
machine-readable outcome.
|
||||||
- Persisted values are fully resolved and immutable.
|
- Persisted values are fully resolved and immutable.
|
||||||
|
|
||||||
The capability is named `humanAvailable`, not `headless`. `headless` describes
|
The capability is named `humanAvailable`, not `headless`. `headless` describes
|
||||||
|
|
|
||||||
|
|
@ -7,6 +7,7 @@ import {
|
||||||
type TurnState,
|
type TurnState,
|
||||||
effectiveTools,
|
effectiveTools,
|
||||||
requestMessagesFor,
|
requestMessagesFor,
|
||||||
|
wrapUpNotice,
|
||||||
} from "@x/shared/dist/turns.js";
|
} from "@x/shared/dist/turns.js";
|
||||||
import type { IContextResolver } from "./context-resolver.js";
|
import type { IContextResolver } from "./context-resolver.js";
|
||||||
|
|
||||||
|
|
@ -46,12 +47,21 @@ export function composeModelRequest(
|
||||||
for (let index = 0; index <= modelCallIndex; index++) {
|
for (let index = 0; index <= modelCallIndex; index++) {
|
||||||
structural.push(...requestMessagesFor(state, index));
|
structural.push(...requestMessagesFor(state, index));
|
||||||
}
|
}
|
||||||
|
// A wrap-up call (final budgeted call of an interactive turn) sends no
|
||||||
|
// tools and appends the budget notice, so the model must answer in text.
|
||||||
|
// Both derive from the durable request flag + config, preserving exact
|
||||||
|
// wire reproducibility.
|
||||||
|
if (call.request.wrapUp) {
|
||||||
|
structural.push(wrapUpNotice(state.definition.config.maxModelCalls));
|
||||||
|
}
|
||||||
return {
|
return {
|
||||||
systemPrompt: agent.systemPrompt,
|
systemPrompt: agent.systemPrompt,
|
||||||
messages: encode(structural),
|
messages: encode(structural),
|
||||||
// The snapshot's base tools plus any durable mid-turn extensions
|
// The snapshot's base tools plus any durable mid-turn extensions
|
||||||
// recorded before this call (tools_extended events).
|
// recorded before this call (tools_extended events).
|
||||||
tools: effectiveTools(state, modelCallIndex, agent.tools),
|
tools: call.request.wrapUp
|
||||||
|
? []
|
||||||
|
: effectiveTools(state, modelCallIndex, agent.tools),
|
||||||
parameters: call.request.parameters,
|
parameters: call.request.parameters,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -1789,6 +1789,81 @@ describe("model-call limit (26.9)", () => {
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Graceful budget wrap-up (issue #768)
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
describe("model-call limit wrap-up", () => {
|
||||||
|
function persistedRequests(log: TEvent[]) {
|
||||||
|
return log.filter(
|
||||||
|
(e): e is Extract<TEvent, { type: "model_call_requested" }> =>
|
||||||
|
e.type === "model_call_requested",
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
it("spends the final budgeted call of a chat turn wrapping up without tools", async () => {
|
||||||
|
const { runtime, repo, models } = makeRuntime({
|
||||||
|
models: [
|
||||||
|
respond(completedResp(assistantCalls(toolCallPart("S", "echo")))),
|
||||||
|
respond(completedResp(assistantText("partial summary"))),
|
||||||
|
],
|
||||||
|
});
|
||||||
|
const turnId = await newTurn(runtime, {
|
||||||
|
config: { humanAvailable: true, maxModelCalls: 2 },
|
||||||
|
});
|
||||||
|
const { outcome } = await advanceAndSettle(runtime, turnId);
|
||||||
|
expect(outcome?.status).toBe("completed");
|
||||||
|
|
||||||
|
// The wrap-up rides the durable request…
|
||||||
|
const requests = persistedRequests(await persisted(repo, turnId));
|
||||||
|
expect(requests[0].request.wrapUp).toBeUndefined();
|
||||||
|
expect(requests[1].request.wrapUp).toBe(true);
|
||||||
|
|
||||||
|
// …and the wire request had no tools plus the budget notice.
|
||||||
|
expect(models.requests[0].tools.length).toBeGreaterThan(0);
|
||||||
|
expect(models.requests[1].tools).toEqual([]);
|
||||||
|
const messages = models.requests[1].messages;
|
||||||
|
const last = messages[messages.length - 1] as { role: string; content: string };
|
||||||
|
expect(last.role).toBe("user");
|
||||||
|
expect(last.content).toMatch(/model-call limit/);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("a 1-call chat budget answers directly as a wrap-up call", async () => {
|
||||||
|
const { runtime, models } = makeRuntime({
|
||||||
|
models: [respond(completedResp(assistantText("hi")))],
|
||||||
|
});
|
||||||
|
const turnId = await newTurn(runtime, {
|
||||||
|
config: { humanAvailable: true, maxModelCalls: 1 },
|
||||||
|
});
|
||||||
|
const { outcome } = await advanceAndSettle(runtime, turnId);
|
||||||
|
expect(outcome?.status).toBe("completed");
|
||||||
|
expect(models.requests[0].tools).toEqual([]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("headless turns keep the hard model-call-limit failure with tools intact", async () => {
|
||||||
|
const { runtime, models } = makeRuntime({
|
||||||
|
models: [
|
||||||
|
respond(completedResp(assistantCalls(toolCallPart("S", "echo")))),
|
||||||
|
respond(completedResp(assistantCalls(toolCallPart("S2", "echo")))),
|
||||||
|
],
|
||||||
|
});
|
||||||
|
const turnId = await newTurn(runtime, {
|
||||||
|
config: {
|
||||||
|
humanAvailable: false,
|
||||||
|
autoPermission: true,
|
||||||
|
maxModelCalls: 2,
|
||||||
|
},
|
||||||
|
});
|
||||||
|
const { outcome } = await advanceAndSettle(runtime, turnId);
|
||||||
|
expect(outcome).toMatchObject({
|
||||||
|
status: "failed",
|
||||||
|
code: MODEL_CALL_LIMIT_ERROR_CODE,
|
||||||
|
});
|
||||||
|
// No wrap-up in headless: the final call was composed with tools.
|
||||||
|
expect(models.requests[1].tools.length).toBeGreaterThan(0);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Tool progress
|
// Tool progress
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
|
||||||
|
|
@ -1190,11 +1190,20 @@ class TurnAdvance {
|
||||||
// what it ran with, and the model bridge maps the canonical value to
|
// what it ran with, and the model bridge maps the canonical value to
|
||||||
// provider-specific options at invoke time.
|
// provider-specific options at invoke time.
|
||||||
const reasoningEffort = this.definition.config.reasoningEffort;
|
const reasoningEffort = this.definition.config.reasoningEffort;
|
||||||
|
// Interactive turns spend their final budgeted call wrapping up: the
|
||||||
|
// composer strips tools and appends the budget notice, so the user
|
||||||
|
// gets a real answer instead of a model-call-limit failure. Headless
|
||||||
|
// turns keep the hard failure — automation needs the machine-readable
|
||||||
|
// outcome, and sub-agents already surface partial results.
|
||||||
|
const wrapUp =
|
||||||
|
this.definition.config.humanAvailable &&
|
||||||
|
index === this.definition.config.maxModelCalls - 1;
|
||||||
const request: z.infer<typeof ModelRequest> = {
|
const request: z.infer<typeof ModelRequest> = {
|
||||||
...(isRef && index === 0 ? { contextRef: context } : {}),
|
...(isRef && index === 0 ? { contextRef: context } : {}),
|
||||||
messages: refs,
|
messages: refs,
|
||||||
parameters:
|
parameters:
|
||||||
reasoningEffort === undefined ? {} : { reasoningEffort },
|
reasoningEffort === undefined ? {} : { reasoningEffort },
|
||||||
|
...(wrapUp ? { wrapUp: true as const } : {}),
|
||||||
};
|
};
|
||||||
await this.append({
|
await this.append({
|
||||||
type: "model_call_requested",
|
type: "model_call_requested",
|
||||||
|
|
|
||||||
|
|
@ -233,8 +233,26 @@ export const ModelRequest = z.object({
|
||||||
contextRef: TurnContextRef.optional(),
|
contextRef: TurnContextRef.optional(),
|
||||||
messages: z.array(ModelRequestMessageRef),
|
messages: z.array(ModelRequestMessageRef),
|
||||||
parameters: z.record(z.string(), z.json()),
|
parameters: z.record(z.string(), z.json()),
|
||||||
|
// Final budgeted call of an interactive turn: the composer strips tools
|
||||||
|
// and appends wrapUpNotice() so the model ends with a real answer
|
||||||
|
// instead of the turn failing with model-call-limit (issue #768).
|
||||||
|
wrapUp: z.literal(true).optional(),
|
||||||
});
|
});
|
||||||
|
|
||||||
|
// The message appended to a wrap-up call's request. A pure function of the
|
||||||
|
// turn's durable config, so replaying the log reproduces the wire bytes.
|
||||||
|
export function wrapUpNotice(
|
||||||
|
maxModelCalls: number,
|
||||||
|
): z.infer<typeof ConversationMessage> {
|
||||||
|
return {
|
||||||
|
role: "user",
|
||||||
|
content:
|
||||||
|
`[system notice] You are on the final model call this turn's budget allows (model-call limit: ${maxModelCalls}); tools are no longer available. ` +
|
||||||
|
"Give your best final answer using only the work above: present what you completed, state clearly what remains unfinished, " +
|
||||||
|
"and let the user know the turn was cut short by the model-call limit, which they can raise in Settings → Advanced.",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
export const ModelCallRequested = z.object({
|
export const ModelCallRequested = z.object({
|
||||||
type: z.literal("model_call_requested"),
|
type: z.literal("model_call_requested"),
|
||||||
turnId: z.string(),
|
turnId: z.string(),
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue