diff --git a/src/controllers/anthropic.js b/src/controllers/anthropic.js index 38178b11..d3038974 100644 --- a/src/controllers/anthropic.js +++ b/src/controllers/anthropic.js @@ -2,6 +2,14 @@ const { isJson, generateUUID } = require('../utils/tools.js'); const { createUsageObject, mergeUpstreamUsage, reportUsage } = require('../utils/precise-tokenizer.js'); const { sendChatRequest, invalidateContextPrefix } = require('../utils/request.js'); const { buildContextPrefixKey } = require('../utils/context-prefix-cache.js'); +const { + gate, + REASONS, + PROTOCOL_RECOVERY_REASONS, + resolveAttemptBudget, + retryHintFor, + appendRetryHint +} = require('../utils/agent-turn-gate.js'); const accountManager = require('../utils/account.js'); const { isChatType, isThinkingEnabled, parserModel, parserMessages, isThinkPhase, extractMediaToFiles, @@ -17,17 +25,13 @@ const { parseToolCallsFromText, createToolCallStreamParser, createNativeToolCallAccumulator, - looksLikeUnexecutedToolAction, containsOrphanProtocolResidue, stripToolCallResidue, - ANSWER_PHASES, - TOOL_CALL_OPEN, - TOOL_CALL_CLOSE + ANSWER_PHASES } = require('../utils/tool-prompt.js'); const { createAgentTagStripper, stripAgentTags, - buildAgentRetryHint, buildAgentTurnDirective, buildToolHistoryLedger, extractHistoryToolCalls, @@ -61,6 +65,7 @@ const { isTransportInterruption, isWafChallengeError, noteRateLimitedAccount, + unclassifiedFailure, RATE_LIMIT_ANTHROPIC_TYPE, UpstreamResponseError } = require('../utils/upstream-error.js'); @@ -124,6 +129,28 @@ const toAnthropicToolUseId = (id) => { return newAnthropicToolUseId(); }; +/** + * Un veredicto de upstream en la forma de error de cable Anthropic, mientras la respuesta + * sigue libre. Es la unica traduccion: la usan la via de excepcion y la via de retorno del + * modulo de request, que antes contestaba un 500 `api_error` para todo. + * @param {object} res - Respuesta HTTP + * @param {{rateLimited: boolean, overloaded: boolean, status: number, retryAfter: number|null}} failure + * @param {string} message - Mensaje para el cliente + */ +const anthropicErrorType = (failure) => failure.rateLimited + ? RATE_LIMIT_ANTHROPIC_TYPE + : (failure.overloaded ? 'overloaded_error' : 'api_error'); + +const writeAnthropicHttpFailure = (res, failure, message) => { + const errorType = anthropicErrorType(failure); + // Retry-After solo con una espera que mando el upstream de verdad. + if (failure.retryAfter !== null) res.set({ 'Retry-After': String(failure.retryAfter) }); + return res.status(failure.status).json({ + type: 'error', + error: { type: errorType, message } + }); +}; + const writeAnthropicError = (res, message, errorType = 'api_error', retryAfterSeconds = null) => { const error = { type: errorType, message }; // A media transmision la cabecera Retry-After ya no se puede poner: el evento es el @@ -958,36 +985,34 @@ const buildInternalRequest = async (anthropicReq) => { }; }; +/** 这一层请求的重试提示词前面带的标题(两处调用点共用同一个字节)。 */ +const ANTHROPIC_RETRY_HINT_HEADER = '# Tool-call retry'; + /** - * 在请求体中追加用于 required 重试的强制提示 - * @param {Object} body - 内部请求体 - * @param {string} hint - 重试提示词 - * @returns {Object} 新请求体 + * La política de esta superficie para la puerta (campos con nombre; el loop pone los hechos, + * la puerta la regla). Los cuatro valores son los de hoy, byte a byte: + * proseWithTools — la prosa junto a llamadas se acepta: los bloques tool_use ya salieron + * discretos y el cliente puede actuar con lo que llegó. + * acceptBareFinal — la prosa sin envoltorio de cierre es una respuesta final: acá los tags + * se pelan sin interpretarse, no hay vocabulario de control que exigir. + * toolErrorsBeforeRequired — `required` manda: si el tool_choice exigía una llamada, el + * rechazo es required_tool (su hint nombra el tool_choice), no tool_error. + * toolErrorsVetoWithCalls — un error de herramienta no veta una llamada que ya se emitió. + * Desde el ticket 06 los dos loops Anthropic leen esta misma constante: no comparten sólo el + * comportamiento, comparten la política. */ -const appendRetryHint = (body, hint) => { - const messages = Array.isArray(body.messages) - ? body.messages.map(message => ({ ...message })) - : []; - if (messages.length === 0) { - messages.push({ role: 'user', content: hint }); - } else { - const last = messages[messages.length - 1]; - if (typeof last.content === 'string') { - last.content = `${last.content}\n\n# Tool-call retry\n${hint}`; - } else if (Array.isArray(last.content)) { - const textPart = last.content.find(part => part?.type === 'text'); - if (textPart) { - textPart.text = `${textPart.text || ''}\n\n# Tool-call retry\n${hint}`; - } else { - last.content = [{ type: 'text', text: hint }, ...last.content]; - } - } - } - return { ...body, messages }; -}; +const ANTHROPIC_GATE_POLICY = Object.freeze({ + proseWithTools: true, + acceptBareFinal: true, + toolErrorsBeforeRequired: false, + toolErrorsVetoWithCalls: false +}); /** - * 判断 tool_choice 是否需要强制调用 + * 判断 tool_choice 是否需要强制调用。 + * No se borra con el `decideRetryReason` del loop no-stream (ticket 06): le quedan llamadores + * fuera de la decisión de turno — el snapshot del loop streaming, y la capa de entrega de las + * dos superficies (`hasToolProtocolError` / el 502 de protocolo). * @param {string|Object} toolChoice - 内部 tool_choice * @returns {boolean} 是否要求至少一次工具调用 */ @@ -997,29 +1022,9 @@ const requiresToolCall = (toolChoice) => { return false; }; -/** - * 构建 required 重试提示 - * @param {string|Object} toolChoice - 内部 tool_choice - * @returns {string} 提示文本 - */ -const buildRetryHint = (toolChoice) => { - if (toolChoice && typeof toolChoice === 'object' && toolChoice.function?.name) { - return `You did not call any tool. You MUST now call \`${toolChoice.function.name}\` using the ${TOOL_CALL_OPEN}...${TOOL_CALL_CLOSE} format.`; - } - return `You did not call any tool. You MUST now call exactly one tool using the ${TOOL_CALL_OPEN}...${TOOL_CALL_CLOSE} format.`; -}; - -const buildEmptyOutputRetryHint = () => [ - 'Your previous reply produced no visible final answer or executable tool call.', - `Continue the Agent task now. If any action remains, emit the required \`${TOOL_CALL_OPEN}\` block immediately with no preamble.`, - 'Only give a normal final answer when the task is actually complete; do not repeat hidden reasoning.' -].join(' '); - -const buildMissingToolRetryHint = () => [ - 'Your previous reply described an action but did not execute any tool call.', - `Perform that action now by emitting the real \`${TOOL_CALL_OPEN}\` block immediately with no preamble.`, - 'Do not describe the action again or claim completion without a tool result.' -].join(' '); +// Los constructores de hint y el append viven sólo en utils/agent-turn-gate.js: el corpus +// grabó esos textos byte a byte y dos copias derivan. Los dos loops de esta superficie piden +// su hint con `retryHintFor(reason, snapshot, context)` — ya no hay copia local que mantener. /** * 把解析器的错误列表压成一行可读的诊断串。 @@ -1045,35 +1050,6 @@ const describeToolErrors = (errors) => { return parts.join('; ') || 'unspecified'; }; -/** - * 工具错误的重试提示。基础文本复用 agent-turn.js 的通用提示;当错误是编造的工具名时, - * 补上真实的名字 —— 那是让这类错误可恢复的唯一信息。原生调用的参数不合法 - * (invalid_arguments / schema_mismatch)时,点名该工具:模型要重发的是参数,不是名字。 - * @param {Array} errors - 本轮的工具错误 - * @param {Array} allowedToolNames - 本次请求真正提供的工具名 - * @returns {string} 提示文本 - */ -const buildToolErrorRetryHint = (errors, allowedToolNames) => { - const base = buildAgentRetryHint('invalid_tool_call'); - const unknown = [...new Set( - errors.filter(e => e?.type === 'unknown_tool').map(e => e.name).filter(Boolean) - )]; - const badArguments = [...new Set( - errors.filter(e => e?.type === 'invalid_arguments' || e?.type === 'schema_mismatch').map(e => e.name).filter(Boolean) - )]; - const lines = [base]; - if (unknown.length && allowedToolNames?.length) { - lines.push( - `The tool name(s) ${unknown.join(', ')} do not exist.`, - `Use ONLY these exact tool names: ${allowedToolNames.join(', ')}.` - ); - } - if (badArguments.length) { - lines.push(`Your arguments for tool ${badArguments.join(', ')} were not a valid JSON object or missed required keys. Re-emit the call with a complete JSON object that matches the tool's input schema.`); - } - return lines.join('\n'); -}; - /** * 异步迭代上游 axios 流,按 SSE 段切分回调内部 delta JSON * @param {object} upstream - axios stream 响应 @@ -1323,9 +1299,10 @@ const handleAnthropicStream = async (res, ctx, upstream) => { let upstreamEventCount; let visibleText = ''; // 本轮 attempt 写到线上的正文。visibleText 是跨轮累计(它如实映照线上已发出的 - // 一切,供 empty 判定和"已见正文只许一次补偿"守卫使用);但 malformed_protocol / - // missing_tool 检查的是**这一轮**说了什么 —— 上一轮泄漏的残渣已经重试过了, - // 拿累计文本判会把成功的重试轮再判一次死。 + // 一切,供"已见正文只许一次补偿"守卫使用);但 malformed_protocol / missing_tool + // 检查的是**这一轮**说了什么 —— 上一轮泄漏的残渣已经重试过了,拿累计文本判会把 + // 成功的重试轮再判一次死。`empty` 自 ticket 05 起 también es de intento: la puerta + // recibe este texto, no el acumulado (medido invisible contra el corpus, ticket 03). let attemptVisibleText = ''; // 本轮 attempt 的**原始**思考文本(不含注入的 searchTable)。think 内容照旧 // verbatim 流给客户端(遏制是另案,见 deferred-work),但回合定案时要拿它过一遍 @@ -1334,7 +1311,8 @@ const handleAnthropicStream = async (res, ctx, upstream) => { // 早有这道防御(openai-agent-runtime.js:232-246);这里把 B 拉到同一水位。 let attemptThinkText = ''; // 思维阶段的排放证据:think 文本过共享解析器后出现调用或解析错误,却没资格 - // 晋升(守卫见回合定案处)。decideRetryReason 据此点起一次性 thought_tool_call。 + // 晋升(守卫见回合定案处)。la puerta la lee como `thinkEvidence` y enciende el + // thought_tool_call de un solo uso. let attemptThinkEvidence = false; // 每个 attempt 都必须拿到全新的解析器。旧代码只建一次,于是补偿重试会继承上一轮的 @@ -1689,71 +1667,15 @@ const handleAnthropicStream = async (res, ctx, upstream) => { ...(nativeToolAccumulator?.getErrors() || []) ]; - /** - * 判断本轮是否需要重试;返回 null 表示接受本轮。 - * 只在 flush 之后调用:flush 会结算挂起的工具调用,此后 hasPendingCall() 恒为假。 - */ - const decideRetryReason = (emittedCalls) => { - if (emittedCalls) return null; - if (parser && requiresToolCall(toolChoice)) return 'required'; - // 以前任何一个工具错误都会让全部补偿失效并直接 502。可是被编造的工具名恰恰是 - // 最容易纠正的错误:把允许的名字摆在模型面前即可。终止性 finish 下**原生来源** - // 的错误不点火:被 length 截断的快照是 truncated_native_call,不发射也不重试 - // (文本来源保持今天的行为)。 - const retryableToolErrors = terminalFinish() ? (parser?.getErrors() || []) : currentToolErrors(); - if (retryableToolErrors.length > 0) return 'tool_error'; - // 平台把模型的原生工具调用吃掉时,我们收到的只剩 role:function 丢弃帧和一段 - // 叙述失败的散文。丢弃帧就是拦截的现场证据:有丢弃、零工具调用、且本请求 - // 确实带工具 → 值得用规范标记提示模型重发一次。终止性 finish(length/ - // content_filter/refusal)与 missing_tool/empty 同一纪律:不重试。 - if (hasTools && normalizeDelta.interceptedToolNames.length > 0 && !terminalFinish()) { - return 'intercepted'; - } - // 同族防御:模型把方括号协议写坏,解析器的抢救闸门也没收下(未知名字 / 缺 - // 闭标记 / 非法 JSON),残渣按正文泄漏。只是重试信号。intercepted 在前—— - // 丢弃帧是更强的证据。判**本轮**文本,不判累计:上一轮的残渣已经重试过了。 - if (hasTools && containsOrphanProtocolResidue(attemptVisibleText) && !terminalFinish()) { - return 'malformed_protocol'; - } - // 同族第三形态:调用(或其残骸)泄漏在 think phase 里,晋升守卫没放行。 - // 排在 missing_tool 之前 —— think 里的排放证据比正文措辞的启发式更硬。 - // 泄漏的调用永远不从这里执行,这只是重试信号。 - if (hasTools && attemptThinkEvidence && !terminalFinish()) { - return 'thought_tool_call'; - } - if (hasTools && looksLikeUnexecutedToolAction(attemptVisibleText) && !terminalFinish()) { - return 'missing_tool'; - } - if (!visibleText.trim() && !terminalFinish()) return 'empty'; - return null; - }; - - const retryHintFor = (reason) => { - let hint; - if (reason === 'required') hint = buildRetryHint(toolChoice); - else if (reason === 'missing_tool') hint = buildMissingToolRetryHint(); - else if (reason === 'empty') hint = buildEmptyOutputRetryHint(); - else if (reason === 'intercepted') hint = buildAgentRetryHint('intercepted'); - else if (reason === 'malformed_protocol') hint = buildAgentRetryHint('malformed_protocol'); - else if (reason === 'thought_tool_call') hint = buildAgentRetryHint('thought_tool_call'); - else hint = buildToolErrorRetryHint(currentToolErrors(), allowedToolNames); - // required / missing_tool 优先级高于 intercepted,会把拦截藏在自己后面。 - // 不动优先级、不动上限——只让提示词把关键事实带上:调用没到客户端。 - if ((reason === 'required' || reason === 'missing_tool') && - normalizeDelta.interceptedToolNames.length > 0) { - hint = `${hint}\n${buildAgentRetryHint('intercepted')}`; - } - // 同一个模式的 think 版本:required / tool_error 盖住 thought_tool_call 时, - // 提示词仍要带上关键事实 —— 调用写在了模型自己够不到的隐藏推理里。 - // (missing_tool / empty 排在 thought_tool_call 之后,证据在时轮不到它们。) - if ((reason === 'required' || reason === 'tool_error') && attemptThinkEvidence) { - hint = `${hint}\n${buildAgentRetryHint('thought_tool_call')}`; - } - return hint; - }; + // El juicio del turno y el armado del hint ya no viven acá: la decisión es + // `gate(snapshot, ANTHROPIC_GATE_POLICY)` y el texto es `retryHintFor(reason, snapshot, + // ...)`, ambos en utils/agent-turn-gate.js (tickets 05 y 06). El loop conserva lo suyo: el + // presupuesto de intentos, la bandera mutable del cupo de recuperación de protocolo y la + // maquinaria de entrega. El snapshot se arma en el punto de liquidación, después del flush + // — flush 会结算挂起的工具调用,此后 hasPendingCall() 恒为假. const config = require('../config/index.js'); - const maxAttempts = Math.max(1, Number(config.agentTurnMaxAttempts) || 1); + const maxAttempts = resolveAttemptBudget(null, config.agentTurnMaxAttempts); let currentUpstream = upstream; // Vueltas en las que el modelo llego a responder. Un failover no cuenta: el modelo aun no @@ -1864,7 +1786,7 @@ const handleAnthropicStream = async (res, ctx, upstream) => { // 1) 必须有非空白名单(无白名单时共享解析器的名字闸门放行一切 —— fail closed, // 不晋升); // 2) 正文侧零工具错误(A 靠 evaluate 先按 toolErrors 拒绝整轮达到同一效果, - // B 的晋升发生在 decideRetryReason 之前,必须自己带上这条)。 + // B 的晋升发生在 la puerta decide 之前,必须自己带上这条)。 // 终止性 finish(length/content_filter/refusal)既不晋升也不重试 —— 与 // intercepted/missing_tool/empty 同一纪律。这不是新的安全边界:A 自兼容工作以来 // 一直在做同一个晋升。守卫不满足但 think 里确实出现了调用(或其解析残骸)时, @@ -1888,8 +1810,30 @@ const handleAnthropicStream = async (res, ctx, upstream) => { } } - const retryReason = decideRetryReason(hasEmittedToolCalls); - if (!retryReason) break; + // Punto de liquidación del turno: un snapshot por intento, una llamada a la puerta. + // `visibleText` es el de **este** intento (el scope de `empty` es attempt-scoped desde + // el ticket 05; medido invisible contra el corpus, ver el ticket 03). `toolCalls` va + // vacío: en esta superficie las llamadas se emiten en el acto y el caso "ya salió un + // bloque tool_use" es `callsDelivered` — un snapshot no puede retractar lo entregado. + const gateSnapshot = { + finishReason: upstreamFinishReason, + visibleText: attemptVisibleText, + controlKind: null, + toolCalls: [], + toolErrors: currentToolErrors(), + textToolErrors: parser ? parser.getErrors() : [], + nativeToolCalls: [], + interceptedToolNames: normalizeDelta.interceptedToolNames, + thinkEvidence: attemptThinkEvidence, + callsDelivered: hasEmittedToolCalls, + textChannelCut: !!textRunaway?.cutRule(), + orphanResidue: containsOrphanProtocolResidue(attemptVisibleText), + hasTools, + requiresToolCall: requiresToolCall(toolChoice) + }; + const decision = gate(gateSnapshot, ANTHROPIC_GATE_POLICY); + if (decision.verdict === 'accept') break; + const retryReason = decision.reason; if (attemptsMade >= maxAttempts) { // 以前这里静默 break:生产环境分不清"回合被接受"和"次数用尽"。措辞保持中立: // 接下来可能按原样交付,也可能收敛成 invalid_tool_call_error / api_error( @@ -1907,9 +1851,7 @@ const handleAnthropicStream = async (res, ctx, upstream) => { // 注意这个上限独立于下面的已见正文守卫 —— 无叙述的拦截(零可见正文)也必须 // 停在一次。放弃时必须留日志:生产环境要能区分"提示被采纳、回合恢复"和 // "第二次、原样交付"。 - const isProtocolRecovery = retryReason === 'intercepted' || - retryReason === 'malformed_protocol' || - retryReason === 'thought_tool_call'; + const isProtocolRecovery = PROTOCOL_RECOVERY_REASONS.has(retryReason); if (isProtocolRecovery && protocolRecoveryRetried) { const giveUpDrops = normalizeDelta.interceptedToolNames.length > 0 ? ` (dropped: ${normalizeDelta.interceptedToolNames.join(', ')})` @@ -1937,7 +1879,7 @@ const handleAnthropicStream = async (res, ctx, upstream) => { // thought_tool_call 消费的同样是这一次"已见正文后的补偿"名额:叙述已经流出 // 去了,但迟到的 tool_use 仍然胜过一个死掉的会话(与 intercepted 同一条道理)。 if (retriedAfterVisibleText) { - if (retryReason === 'tool_error') { + if (retryReason === REASONS.TOOL_ERROR) { // 以前这里静默 break:生产环境看不见"本轮是垃圾、按原样交付"的定案。 logger.warn( `Anthropic Agent 已见正文后再次 tool_error,补偿名额已用,按原样交付 (${describeToolErrors(currentToolErrors())})`, @@ -1952,7 +1894,7 @@ const handleAnthropicStream = async (res, ctx, upstream) => { // 拦在 emit 层,检测记账照旧),失败就按今天交付。绝不新增名额;模型复述 // 协议的老毛病(回显字面标签必然解析失败)因此不会把第二轮垃圾拼上线 —— // 垃圾轮的文本根本不上线。 - if (retryReason === 'tool_error') { + if (retryReason === REASONS.TOOL_ERROR) { suppressAttemptOutput = true; // attempt 侧的 recovered 文本进银行(剥掉登记残渣后),交付段仍会交付它。 bankedRecoveredText += stripRecoveredResidue(recoveredBuffer, parser ? parser.getResidueSpans() : []); @@ -1978,7 +1920,13 @@ const handleAnthropicStream = async (res, ctx, upstream) => { let retryResp = null; try { await withPing(async () => { - retryResp = await sendRequest(appendRetryHint(requestBody, retryHintFor(retryReason)), upstreamOptions); + // El hint sale del snapshot ya armado: los errores y la evidencia son los del + // intento que la puerta acaba de rechazar, no los de un instante posterior. + const hint = retryHintFor(retryReason, gateSnapshot, { toolChoice, allowedToolNames }); + retryResp = await sendRequest( + appendRetryHint(requestBody, hint, { header: ANTHROPIC_RETRY_HINT_HEADER }), + upstreamOptions + ); }); } catch (e) { logger.error('Anthropic 流式重试失败', 'ANTHROPIC', '', e); @@ -2369,7 +2317,7 @@ const handleAnthropicNonStream = async (res, ctx, upstream) => { ...(nativeToolAccumulator?.getErrors() || []) ]; // 本轮 parser 的**原始** cleanedText 与登记 span(位置坐标系 = 原始文本)。 - // 检测(decideRetryReason / settleThinkPhase)继续吃 tag-stripped 的 + // 检测(la puerta, vía el snapshot / settleThinkPhase)继续吃 tag-stripped 的 // cleanedText,逐字节不变;剥残渣只在交付点、在原始文本上按位置进行,然后 // 才剥 agent tag(与 B 同序 —— review loop 1,条目 6)。 let roundRawCleanedText = parsedTools.cleanedText; @@ -2415,38 +2363,47 @@ const handleAnthropicNonStream = async (res, ctx, upstream) => { }; settleThinkPhase(); - const decideRetryReason = () => { - if (toolCalls.length > 0) return null; - if (hasTools && requiresToolCall(toolChoice)) return 'required'; - // 以前任何一个工具错误都会让全部补偿失效并直接 502。被编造的工具名恰恰是最容易 - // 纠正的错误:把允许的名字摆在模型面前即可。终止性 finish 下原生来源的错误不点火 - // (截断的快照 = truncated_native_call,不发射也不重试;文本来源保持今天的行为)。 - if ((terminalFinish() ? textToolErrors : toolErrors).length > 0) return 'tool_error'; - // 与流式分支同一条防御:role:function 丢弃帧 + 零工具调用 + 本请求带工具, - // 说明平台吃掉了模型的原生调用,用规范标记提示重发一次。终止性 finish 不重试 - // —— 与 missing_tool/empty 同一纪律。 - if (hasTools && normalizeDelta.interceptedToolNames.length > 0 && !terminalFinish()) { - return 'intercepted'; - } - // 同族防御:方括号协议写坏(孤儿闭标记 / 开头裸负载)整段泄漏为可见正文。 - // 只是重试信号,泄漏的 JSON 永远不执行。intercepted 在前——丢弃帧是更强的证据。 - if (hasTools && containsOrphanProtocolResidue(cleanedText) && !terminalFinish()) { - return 'malformed_protocol'; - } - // 同族第三形态:调用(或其残骸)泄漏在 think phase 里,晋升守卫没放行。 - // 排在 missing_tool 之前;泄漏的调用永远不从这里执行,这只是重试信号。 - if (hasTools && attemptThinkEvidence && !terminalFinish()) { - return 'thought_tool_call'; - } - if (hasTools && looksLikeUnexecutedToolAction(cleanedText) && !terminalFinish()) { - return 'missing_tool'; - } - if (!cleanedText.trim() && !terminalFinish()) return 'empty'; - return null; + // Punto de liquidación del turno: un snapshot por intento, una llamada a la puerta — la + // misma decisión que toma el loop streaming (ticket 06). Acá ya no vive ninguna regla: el + // vocabulario de razones, la precedencia y el texto del hint son de + // utils/agent-turn-gate.js, y ANTHROPIC_GATE_POLICY es la política de las dos superficies. + // + // `callsDelivered` es false: nada salió al cliente todavía y todo intento es retractable — + // ese campo es exactamente lo que hace que este loop y el streaming sean la misma decisión. + // `visibleText` es el de **este** intento (cleanedText se resuelve por ronda, tag-stripped); + // de ahí salen también las dos detecciones que el loop hacía a mano — residuo huérfano de + // protocolo y prosa que narra una acción sin ejecutarla. + // + // Los dos canales de error viajan separados: `textToolErrors` es el subconjunto del parser + // de texto (con finish terminal la puerta decide sólo con él, como decidía este loop — el + // snapshot truncado del acumulador nativo no veta) y `toolErrors` es la unión con los del + // acumulador nativo. Un error de herramienta sigue siendo corregible en vez de 502: el hint + // `tool_error` pone los nombres permitidos delante del modelo. + let gateSnapshot = null; + let decision = null; + const settleTurn = () => { + gateSnapshot = { + finishReason: upstreamFinishReason, + visibleText: cleanedText, + controlKind: null, + toolCalls, + toolErrors, + textToolErrors, + nativeToolCalls, + interceptedToolNames: normalizeDelta.interceptedToolNames, + thinkEvidence: attemptThinkEvidence, + callsDelivered: false, + textChannelCut: !!textRunaway?.cutRule(), + orphanResidue: containsOrphanProtocolResidue(cleanedText), + hasTools, + requiresToolCall: requiresToolCall(toolChoice) + }; + decision = gate(gateSnapshot, ANTHROPIC_GATE_POLICY); }; + settleTurn(); const config = require('../config/index.js'); - const maxAttempts = Math.max(1, Number(config.agentTurnMaxAttempts) || 1); + const maxAttempts = resolveAttemptBudget(null, config.agentTurnMaxAttempts); let attemptsMade = 1; let streamBrokeOnRetry = false; let protocolRecoveryRetried = false; @@ -2455,17 +2412,15 @@ const handleAnthropicNonStream = async (res, ctx, upstream) => { // "迟到的叙述胜过死掉的会话"的精神)。留底形态:{ stripped, raw, spans }。 let narrationFallback = null; - while (attemptsMade < maxAttempts) { - const retryReason = decideRetryReason(); - if (!retryReason) break; + while (attemptsMade < maxAttempts && decision.verdict === 'retry') { + const retryReason = decision.reason; // 与流式分支同一条纪律:协议恢复重试(intercepted / malformed_protocol / // thought_tool_call 共享同一个名额)整个请求只允许一次。第二次说明提示没被 // 采纳,把叙述散文按正常回答交付,别再烧尝试次数。放弃时留日志:生产环境 - // 要能区分"提示被采纳、回合恢复"和"第二次、原样交付"。 - const isProtocolRecovery = retryReason === 'intercepted' || - retryReason === 'malformed_protocol' || - retryReason === 'thought_tool_call'; + // 要能区分"提示被采纳、回合恢复"和"第二次、原样交付"。La regla vive en la + // puerta y es la misma constante en los dos loops (ticket 06). + const isProtocolRecovery = PROTOCOL_RECOVERY_REASONS.has(retryReason); if (isProtocolRecovery) { if (protocolRecoveryRetried) { const giveUpDrops = normalizeDelta.interceptedToolNames.length > 0 @@ -2490,26 +2445,12 @@ const handleAnthropicNonStream = async (res, ctx, upstream) => { 'ANTHROPIC' ); - let hint = retryReason === 'required' - ? buildRetryHint(toolChoice) - : (retryReason === 'missing_tool' - ? buildMissingToolRetryHint() - : (retryReason === 'empty' - ? buildEmptyOutputRetryHint() - : (retryReason === 'intercepted' || retryReason === 'malformed_protocol' || retryReason === 'thought_tool_call' - ? buildAgentRetryHint(retryReason) - : buildToolErrorRetryHint(toolErrors, allowedToolNames)))); - // required / missing_tool 优先级高于 intercepted,会把拦截藏在自己后面。 - // 不动优先级、不动上限——只让提示词把关键事实带上:调用没到客户端。 - if ((retryReason === 'required' || retryReason === 'missing_tool') && - normalizeDelta.interceptedToolNames.length > 0) { - hint = `${hint}\n${buildAgentRetryHint('intercepted')}`; - } - // 同一个模式的 think 版本:required / tool_error 盖住 thought_tool_call 时, - // 提示词仍要带上关键事实 —— 调用写在了模型自己够不到的隐藏推理里。 - if ((retryReason === 'required' || retryReason === 'tool_error') && attemptThinkEvidence) { - hint = `${hint}\n${buildAgentRetryHint('thought_tool_call')}`; - } + // El hint sale del snapshot ya armado — el mismo que la puerta acaba de juzgar, no un + // estado posterior. `retryHintFor` es también quien agrega los dos apéndices que antes se + // pegaban a mano acá: required/missing_tool no esconden la interceptación (el hint lleva + // el hecho "la llamada no llegó"), y required/tool_error no esconden la evidencia de think + // (el hint lleva el hecho "quedó escrita en la razón oculta"). + const hint = retryHintFor(retryReason, gateSnapshot, { toolChoice, allowedToolNames }); // finding 2 的教义对 thought_tool_call 同样成立:14:08 形态(think 泄漏 + 成功 // 叙述)的重试若空手而归,绝不能拿 502 换掉已经拿到的叙述。malformed_protocol @@ -2523,7 +2464,10 @@ const handleAnthropicNonStream = async (res, ctx, upstream) => { let retryResp; try { - retryResp = await sendRequest(appendRetryHint(requestBody, hint), upstreamOptions); + retryResp = await sendRequest( + appendRetryHint(requestBody, hint, { header: ANTHROPIC_RETRY_HINT_HEADER }), + upstreamOptions + ); } catch (e) { logger.error('Anthropic 非流式重试失败', 'ANTHROPIC', '', e); if (e.publicMessage) throw e; @@ -2545,7 +2489,8 @@ const handleAnthropicNonStream = async (res, ctx, upstream) => { // 每轮新建;统一两个循环的计划在 lohari 仓库 // _bmad-output/implementation-artifacts/spec-qwen2api-unify-agent-loop.md)。 // 在那之前:拦截计数必须按轮**就地**归零(length = 0,不能重新赋值 —— - // decideRetryReason 闭包持有的是同一个数组引用),否则上一轮的丢弃会把 + // el normalizador empuja sobre ESE array y el snapshot lo lee de la misma + // propiedad; reasignarla los desincroniza),否则上一轮的丢弃会把 // 成功的重试再判成拦截,协议恢复名额被烧光后以 502 收场。 normalizeDelta.interceptedToolNames.length = 0; // 判定输入按轮清零(thinkingContent 本身继续累计 —— 响应交付语义不动)。 @@ -2572,21 +2517,23 @@ const handleAnthropicNonStream = async (res, ctx, upstream) => { // 本轮文本,残渣原样上线 —— 有测试钉住)。 roundRawCleanedText = parsedRetry.cleanedText; roundResidueSpans = parsedRetry.residueSpans || []; - // 重试轮的 think phase 同样要定案:晋升或留证据,下一次 decideRetryReason 才看得见。 + // 重试轮的 think phase 同样要定案:晋升或留证据,下一轮定案才看得见。 settleThinkPhase(); + // La ronda del reintento también es un intento: se arma su snapshot y la puerta vuelve + // a decidir (el `decision` que lee la condición del loop es este). + settleTurn(); } // 与流式分支对称的收尾观测:次数用尽而最后一轮仍被拒绝时留痕(协议恢复的 // give-up 在循环内已有自己的日志,且只在 attemptsMade < maxAttempts 时触发, // 不会与这行重复)。措辞中立:接下来可能按原样交付、502 或兜底叙述,不预判。 - if (!streamBrokeOnRetry && attemptsMade >= maxAttempts) { - const finalRejection = decideRetryReason(); - if (finalRejection) { - logger.warn( - `Anthropic 非流式 Agent 尝试次数用尽(${attemptsMade}/${maxAttempts}),最后一轮仍被拒绝 (${finalRejection})`, - 'ANTHROPIC' - ); - } + // La razón es la de la última liquidación — el mismo veredicto que dejó al loop sin + // presupuesto, no uno recalculado sobre un estado que ya no cambió. + if (!streamBrokeOnRetry && attemptsMade >= maxAttempts && decision.verdict === 'retry') { + logger.warn( + `Anthropic 非流式 Agent 尝试次数用尽(${attemptsMade}/${maxAttempts}),最后一轮仍被拒绝 (${decision.reason})`, + 'ANTHROPIC' + ); } if (streamBrokeOnRetry) { @@ -2608,7 +2555,7 @@ const handleAnthropicNonStream = async (res, ctx, upstream) => { // salvage-3 layer 3:交付轮登记过残渣才动交付文本(review loop 1,条目 9: // 门挂在 residueSpans 上,不挂 toolErrors —— narrationFallback 轮零错误也可能 // 携带残渣)。位置驱动:在**原始**文本上按登记落点剥,再剥 agent tag(与 B - // 同序)。检测与重试判定(decideRetryReason / containsOrphanProtocolResidue) + // 同序)。检测与重试判定(la puerta / containsOrphanProtocolResidue) // 早已在未剥离文本上跑完 —— 剥离只发生在交付点。剥离必须在下面的空判据 // **之前**(review loop 2):一整轮只有 debris 残渣(无信封负载配不平 —— // 有登记、零 toolErrors)时,剥后为空要走「无正文」的 502,绝不能交付 @@ -2768,10 +2715,14 @@ const handleAnthropicMessages = async (req, res) => { upstreamResp = await sendChatRequest(body, upstreamOptions); currentAccount = upstreamResp.currentAccount || null; if (!upstreamResp.status || !upstreamResp.response) { - return res.status(500).json({ - type: 'error', - error: { type: 'api_error', message: upstreamResp.message || 'Request failed' } - }); + // El fallo dice por que fallo: cuota 429 `rate_limit_error`, sobrecarga 529 + // `overloaded_error`, transporte 503, sin clasificar 502 `api_error`. Nunca un + // `api_error` 500 para todo: eso es indistinguible de "servidor roto". + return writeAnthropicHttpFailure( + res, + upstreamResp.failure || unclassifiedFailure(502), + upstreamResp.message || 'Upstream did not produce a usable response' + ); } // Aviso al cliente cuando el contexto se recortó en silencio. El fallback por fallo @@ -2806,10 +2757,10 @@ const handleAnthropicMessages = async (req, res) => { // La cuota diaria agotada es 429 `rate_limit_error`, como la API nativa — no un 500 // `api_error`. Gemelo: chat.js#writeOpenAIHttpError. La deteccion es unica // (utils/upstream-error.js#describeUpstreamFailure); aqui solo se traduce al cable. + // 500 por defecto: lo que llega aqui son excepciones nuestras o challenges ya + // clasificados, no un no-200 opaco del upstream (ese sale por la via de retorno). const failure = describeUpstreamFailure(error, 500); - const errorType = failure.rateLimited - ? RATE_LIMIT_ANTHROPIC_TYPE - : (failure.overloaded ? 'overloaded_error' : 'api_error'); + const errorType = anthropicErrorType(failure); // La otra mitad: sin esto el cliente deja de reintentar pero el servidor sigue // devolviendo la misma cuenta agotada al sorteo, y la quema en cada vuelta. // Si el failover a mitad de stream ya paso la cuenta a cooldown (recordFailedAccount) @@ -2831,12 +2782,8 @@ const handleAnthropicMessages = async (req, res) => { invalidateContextPrefix(contextPrefixKey); } if (!res.headersSent) { - // Retry-After solo con una espera que mando el upstream de verdad. - if (failure.retryAfter !== null) res.set({ 'Retry-After': String(failure.retryAfter) }); - res.status(failure.status).json({ - type: 'error', - error: { type: errorType, message: error.publicMessage || 'Service error' } - }); + // Mismo escritor que la via de retorno: una sola traduccion al cable Anthropic. + writeAnthropicHttpFailure(res, failure, error.publicMessage || 'Service error'); } else { // A media transmision el status ya no se puede cambiar: el `type` del evento es el // unico canal que le queda al cliente para distinguir cuota de averia. @@ -2864,6 +2811,5 @@ module.exports = { runWithAnthropicPing, handleAnthropicStream, handleAnthropicNonStream, - describeToolErrors, - buildToolErrorRetryHint + describeToolErrors }; diff --git a/src/controllers/chat.js b/src/controllers/chat.js index 0e8a548b..7073cf08 100644 --- a/src/controllers/chat.js +++ b/src/controllers/chat.js @@ -3,32 +3,27 @@ const { createUsageObject, mergeUpstreamUsage, reportUsage } = require('../utils const { sendChatRequest } = require('../utils/request.js') const { buildContextPrefixKey } = require('../utils/context-prefix-cache.js') const { - createToolCallStreamParser, - parseToolCallsFromText, - createNativeToolCallAccumulator, - looksLikeUnexecutedToolAction, stripToolCallResidue, - TOOL_CALL_OPEN, - TOOL_CALL_CLOSE + TOOL_CALL_OPEN } = require('../utils/tool-prompt.js') const { stripAgentTags } = require('../utils/agent-turn.js') const { consumeSSEStream, createUpstreamResponseFilter } = require('../utils/sse.js') const accountManager = require('../utils/account.js') const config = require('../config/index.js') const { logger } = require('../utils/logger') -const { createUpstreamDeltaNormalizer, createClientToolNamePredicate } = require('../utils/chat-helpers.js') +const { createUpstreamDeltaNormalizer } = require('../utils/chat-helpers.js') const { assertNoUpstreamFailure, describeUpstreamFailure, isRateLimitError, isWafChallengeError, noteRateLimitedAccount, - RATE_LIMIT_OPENAI_TYPE + unclassifiedFailure, + openAIErrorShape } = require('../utils/upstream-error.js') -const { runOpenAIAgentTurn, feedNativeFrame } = require('../utils/openai-agent-runtime.js') +const { runOpenAIAgentTurn } = require('../utils/openai-agent-runtime.js') -const normalizeOpenAIFinishReason = (upstreamReason, hasToolCalls, upstreamCompleted) => { - if (hasToolCalls) return 'tool_calls' +const normalizeOpenAIFinishReason = (upstreamReason, upstreamCompleted) => { if (typeof upstreamReason === 'string' && upstreamReason.length > 0) { const aliases = { end_turn: 'stop', @@ -93,43 +88,12 @@ const getImageMarkdownListFromDelta = (delta) => { return imageList } -/** - * 判断 tool_choice 是否要求强制调用工具 - * @param {string|Object} toolChoice - OpenAI tool_choice - * @returns {boolean} 是否需要至少一次工具调用 - */ -const requiresToolCall = (toolChoice) => { - if (toolChoice === 'required') return true - if (toolChoice && typeof toolChoice === 'object' && toolChoice.type === 'function' && toolChoice.function?.name) { - return true - } - return false -} - -/** - * 构建 tool_choice=required 重试时追加的强约束提示 - * @param {string|Object} toolChoice - OpenAI tool_choice - * @returns {string} 重试提示词 - */ -const buildRequiredRetryHint = (toolChoice) => { - if (toolChoice && typeof toolChoice === 'object' && toolChoice.function?.name) { - return `You did not call any tool in your previous reply. You MUST now call the tool \`${toolChoice.function.name}\` using the ${TOOL_CALL_OPEN}...${TOOL_CALL_CLOSE} format and nothing else.` - } - return `You did not call any tool in your previous reply. You MUST now call exactly one tool using the ${TOOL_CALL_OPEN}...${TOOL_CALL_CLOSE} format and nothing else.` -} - const buildEmptyOutputRetryHint = () => [ 'Your previous reply produced no visible final answer or executable tool call.', `Continue the Agent task now. If any action remains, emit the required \`${TOOL_CALL_OPEN}\` block immediately with no preamble.`, 'Only give a normal final answer when the task is actually complete; do not repeat hidden reasoning.' ].join(' ') -const buildMissingToolRetryHint = () => [ - 'Your previous reply described an action but did not execute any tool call.', - `Perform that action now by emitting the real \`${TOOL_CALL_OPEN}\` block immediately with no preamble.`, - 'Do not describe the action again or claim completion without a tool result.' -].join(' ') - const appendRetryHintToRequestBody = (requestBody, hint) => { const messages = Array.isArray(requestBody?.messages) ? requestBody.messages.map(message => ({ ...message })) @@ -160,7 +124,7 @@ const appendRetryHintToRequestBody = (requestBody, hint) => { * @param {boolean} enable_web_search - 是否启用网络搜索 * @param {object} requestBody - 原始请求体,用于提取prompt信息 * @param {object} [options] - 扩展选项 - * @param {boolean} [options.has_tools] - 是否启用工具调用解析 + * @param {boolean} [options.has_tools] - 是否走 Agent 工具处理器(是则不进入本函数) * @param {string|Object} [options.tool_choice] - OpenAI tool_choice 控制项 */ /** @@ -225,18 +189,13 @@ const writeOpenAIHttpError = (res, error = {}) => { */ const upstreamErrorShape = (error, fallbackMessage, fallbackCode = 'upstream_error') => { // 529 es un status de Anthropic; en el cable OpenAI el adjunto caido es 503. - const failure = describeUpstreamFailure(error, 502, 503) - const shape = { - status: failure.status, - message: error?.publicMessage || fallbackMessage, - code: failure.rateLimited - ? RATE_LIMIT_OPENAI_TYPE - : (failure.overloaded ? 'upstream_unavailable' : (error?.code || fallbackCode)) - } - if (failure.rateLimited) shape.type = RATE_LIMIT_OPENAI_TYPE - else if (failure.overloaded) shape.type = 'server_error' - if (failure.retryAfter !== null) shape.retry_after = failure.retryAfter - return shape + // La traduccion al cable vive en utils/upstream-error.js#openAIErrorShape, junto a los dos + // vocabularios: la comparten esta via, la de retorno y el runtime de agente. + return openAIErrorShape( + describeUpstreamFailure(error, 502, 503), + error?.publicMessage || fallbackMessage, + error?.code || fallbackCode + ) } const runWithProcessingHeartbeat = async (res, work, intervalMs = 15000) => { @@ -630,7 +589,7 @@ const handleOpenAIAgentNonStream = async ( } const handleStreamResponse = async (res, response, enable_thinking, enable_web_search, requestBody = null, options = {}) => { - if (options.has_tools && options.strict_agent_turn !== false) { + if (options.has_tools) { return handleOpenAIAgentStream( res, response, @@ -650,18 +609,7 @@ const handleStreamResponse = async (res, response, enable_thinking, enable_web_s let emittedImageMarkdownSet = new Set() let pendingImageMarkdownList = [] - const hasTools = !!options.has_tools const requestSender = options.sendChatRequest || sendChatRequest - const toolChoice = options.tool_choice - const allowedToolNames = options.allowed_tool_names || [] - const isClientToolName = createClientToolNamePredicate(allowedToolNames) - let toolParser = hasTools ? createToolCallStreamParser({ allowedToolNames }) : null - let nativeToolAccumulator = hasTools - ? createNativeToolCallAccumulator({ allowedToolNames }) - : null - // 调用方持有唯一的单调 index:文本解析器与原生累积器各自从 0 计数,直接透传会让 - // 两路都写 tool_calls[0]。 - let nextToolCallIndex = 0 let upstreamFinishReason = null let upstreamCompleted = false let upstreamEventCount = 0 @@ -730,86 +678,6 @@ const handleStreamResponse = async (res, response, enable_thinking, enable_web_s })}\n\n`) } - /** - * 发送回复正文增量:有工具解析器则先过解析,否则直接写 content - * @param {string} text - 回复正文文本 - */ - const emitAnswerContent = (text) => { - if (!text) return - if (toolParser) { - const parsed = toolParser.push(text) - if (parsed.textDelta) writeContentDelta(parsed.textDelta) - if (parsed.completedCalls.length > 0) writeToolCallsDelta(parsed.completedCalls) - } else { - writeContentDelta(text) - } - } - - /** - * 写一个工具调用增量,按 OpenAI 规范分片: - * 1) 头块:包含 index/id/type 与 function.name + 空 arguments - * 2) 多个参数块:function.arguments 切片 - * @param {Array} calls - 已完成的工具调用列表 - */ - const writeToolCallsDelta = (calls) => { - if (!calls || calls.length === 0) return - const ARG_CHUNK_SIZE = 32 - - for (const call of calls) { - const index = nextToolCallIndex++ - const headerDelta = { - "id": `chatcmpl-${message_id}`, - "object": "chat.completion.chunk", - "created": Math.round(new Date().getTime() / 1000), - "choices": [ - { - "index": 0, - "delta": { - "tool_calls": [ - { - "index": index, - "id": call.id, - "type": "function", - "function": { - "name": call.function.name, - "arguments": "" - } - } - ] - }, - "finish_reason": null - } - ] - } - res.write(`data: ${JSON.stringify(headerDelta)}\n\n`) - - const argsString = call.function.arguments || '' - for (let offset = 0; offset < argsString.length; offset += ARG_CHUNK_SIZE) { - const piece = argsString.slice(offset, offset + ARG_CHUNK_SIZE) - const argDelta = { - "id": `chatcmpl-${message_id}`, - "object": "chat.completion.chunk", - "created": Math.round(new Date().getTime() / 1000), - "choices": [ - { - "index": 0, - "delta": { - "tool_calls": [ - { - "index": index, - "function": { "arguments": piece } - } - ] - }, - "finish_reason": null - } - ] - } - res.write(`data: ${JSON.stringify(argDelta)}\n\n`) - } - } - } - /** * 处理一个 SSE data 段(已剥离 'data: ' 前缀) * @param {string} dataContent - 原始 data 段 @@ -833,11 +701,6 @@ const handleStreamResponse = async (res, response, enable_thinking, enable_web_s } const delta = choice.delta || {} - if (nativeToolAccumulator) { - // 关闭即判定;发射仍在回合尾部 finalize()(旧路径没有中途排放,drain 为空操作)。 - feedNativeFrame(nativeToolAccumulator, delta, reportedFinishReason, { isClientToolName, drain: () => {} }) - } - if (delta && delta.name === 'web_search') { web_search_info = delta.extra.web_search_info } @@ -893,13 +756,7 @@ const handleStreamResponse = async (res, response, enable_thinking, enable_web_s } } - if (toolParser && delta.phase === 'answer') { - const parsed = toolParser.push(content) - if (parsed.textDelta) writeContentDelta(parsed.textDelta) - if (parsed.completedCalls.length > 0) writeToolCallsDelta(parsed.completedCalls) - } else { - writeContentDelta(content) - } + writeContentDelta(content) return } @@ -927,7 +784,7 @@ const handleStreamResponse = async (res, response, enable_thinking, enable_web_s content = `${pendingImageContent}${content}` } } - emitAnswerContent(content) + writeContentDelta(content) } /** @@ -951,52 +808,22 @@ const handleStreamResponse = async (res, response, enable_thinking, enable_web_s await pipeUpstream(response) - // Agent 空回合补偿:只有思考、没有正文/工具调用时自动重试一次。 - // required 仍使用更强的指定工具提示;两个条件共用一次重试,避免重复请求。 - const needsRequiredRetry = !!( - hasTools && toolParser && - !toolParser.hasEmittedAnyCall() && - !nativeToolAccumulator?.hasAny() && - requiresToolCall(toolChoice) - ) + // 空回合补偿:只有思考、没有正文时自动重试一次。 + // Con herramientas la peticion no llega aqui — entra por el handler de agente en el + // despacho de arriba — asi que este camino ya no tiene parser ni acumulador que + // reconstruir por intento. const needsEmptyOutputRetry = !!( !visibleContent.trim() && - !toolParser?.hasEmittedAnyCall() && - !toolParser?.hasPendingCall() && - !toolParser?.hasParseError() && - !nativeToolAccumulator?.hasAny() && - !['length', 'max_tokens', 'content_filter', 'refusal'].includes(upstreamFinishReason) - ) - const needsMissingToolRetry = !!( - hasTools && looksLikeUnexecutedToolAction(visibleContent) && - !toolParser?.hasEmittedAnyCall() && !toolParser?.hasPendingCall() && - !toolParser?.hasParseError() && !nativeToolAccumulator?.hasAny() && !['length', 'max_tokens', 'content_filter', 'refusal'].includes(upstreamFinishReason) ) - if (needsRequiredRetry || needsEmptyOutputRetry || needsMissingToolRetry) { - const retryHint = needsRequiredRetry - ? buildRequiredRetryHint(toolChoice) - : (needsMissingToolRetry ? buildMissingToolRetryHint() : buildEmptyOutputRetryHint()) - const retryBody = appendRetryHintToRequestBody(requestBody, retryHint) - logger.warn( - needsRequiredRetry - ? 'tool_choice=required 首次未触发工具调用,进行一次重试' - : (needsMissingToolRetry - ? 'Agent 首次响应只描述了动作但未调用工具,进行一次补偿重试' - : 'Agent 首次响应没有正文或工具调用,进行一次补偿重试'), - 'CHAT' - ) + if (needsEmptyOutputRetry) { + const retryBody = appendRetryHintToRequestBody(requestBody, buildEmptyOutputRetryHint()) + logger.warn('Agent 首次响应没有正文或工具调用,进行一次补偿重试', 'CHAT') try { // Mismas opciones que la peticion original: sin ellas el reenvio no puede // compactar ni reutilizar el prefijo de historial y quema un parse mas. const retryResp = await requestSender(retryBody, options.upstreamOptions || {}) if (retryResp.status && retryResp.response) { - // 与非流式分支同一条:重试是新的回合,解析器与累积器都重建,第一轮的残片 - // 不能漂进第二轮(其余消费者本来就按 attempt 重建)。 - if (hasTools) { - toolParser = createToolCallStreamParser({ allowedToolNames }) - nativeToolAccumulator = createNativeToolCallAccumulator({ allowedToolNames }) - } upstreamFinishReason = null await pipeUpstream(retryResp.response) } @@ -1006,44 +833,13 @@ const handleStreamResponse = async (res, response, enable_thinking, enable_web_s } } - // flush 工具调用解析器中的残留内容 - if (toolParser) { - const tail = toolParser.flush() - if (tail.textDelta) writeContentDelta(tail.textDelta) - if (tail.completedCalls.length > 0) writeToolCallsDelta(tail.completedCalls) - } - - const nativeToolCalls = nativeToolAccumulator?.hasAny() - ? nativeToolAccumulator.finalize() - : [] - if (nativeToolCalls.length > 0) writeToolCallsDelta(nativeToolCalls) - - const hasEmittedToolCalls = !!( - nativeToolCalls.length > 0 || - (toolParser && toolParser.hasEmittedAnyCall()) - ) - const hasToolProtocolError = !!( - !hasEmittedToolCalls && - (requiresToolCall(toolChoice) || - (toolParser && toolParser.hasParseError()) || - (nativeToolAccumulator && nativeToolAccumulator.hasParseError())) - ) - if (hasToolProtocolError) { - writeOpenAIStreamError(res, '上游返回了残缺、非法或不存在的工具调用', 'invalid_tool_call') - return - } - - if (!visibleContent.trim() && !hasEmittedToolCalls && + if (!visibleContent.trim() && !['length', 'max_tokens', 'content_filter', 'refusal'].includes(upstreamFinishReason)) { writeOpenAIStreamError(res, '上游重试后仍未返回正文或工具调用', 'upstream_empty_output') return } - const finishReason = normalizeOpenAIFinishReason( - upstreamFinishReason, - hasEmittedToolCalls, - upstreamCompleted - ) + const finishReason = normalizeOpenAIFinishReason(upstreamFinishReason, upstreamCompleted) if (!finishReason) { const detail = upstreamEventCount === 0 ? '上游未返回任何 SSE 事件' : '上游流在结束标记前断开' writeOpenAIStreamError(res, detail, 'upstream_incomplete') @@ -1126,10 +922,10 @@ const handleStreamResponse = async (res, response, enable_thinking, enable_web_s * @param {string} model - 模型名称 * @param {object} requestBody - 原始请求体,用于提取prompt信息 * @param {object} [options] - 扩展选项 - * @param {boolean} [options.has_tools] - 是否启用工具调用解析 + * @param {boolean} [options.has_tools] - 是否走 Agent 工具处理器(是则不进入本函数) */ const handleNonStreamResponse = async (res, response, enable_thinking, enable_web_search, model, requestBody = null, options = {}) => { - if (options.has_tools && options.strict_agent_turn !== false) { + if (options.has_tools) { return handleOpenAIAgentNonStream( res, response, @@ -1151,14 +947,7 @@ const handleNonStreamResponse = async (res, response, enable_thinking, enable_we let appendedImageMarkdownSet = new Set() let pendingImageMarkdownList = [] - const hasTools = !!options.has_tools const requestSender = options.sendChatRequest || sendChatRequest - const toolChoice = options.tool_choice - const allowedToolNames = options.allowed_tool_names || [] - const isClientToolName = createClientToolNamePredicate(allowedToolNames) - let nativeToolAccumulator = hasTools - ? createNativeToolCallAccumulator({ allowedToolNames }) - : null let upstreamFinishReason = null let upstreamCompleted = false let upstreamEventCount = 0 @@ -1206,11 +995,6 @@ const handleNonStreamResponse = async (res, response, enable_thinking, enable_we upstreamFinishReason = reportedFinishReason } const delta = choice.delta || {} - if (nativeToolAccumulator) { - // 关闭即判定;结算在回合尾部 finalize()(drain 为空操作)。 - feedNativeFrame(nativeToolAccumulator, delta, reportedFinishReason, { isClientToolName, drain: () => {} }) - } - if (delta.name === 'web_search') { web_search_info = delta.extra?.web_search_info } @@ -1292,49 +1076,27 @@ const handleNonStreamResponse = async (res, response, enable_thinking, enable_we }) } - // 同时支持提示词/XML 工具调用与上游原生 delta.tool_calls。 + // Sin herramientas no hay protocolo que parsear: lo acumulado es la respuesta tal + // cual. Las peticiones con tools entran por el handler de agente, en el despacho de + // arriba, y nunca llegan a esta funcion. let assistantContent = fullContent - let toolCalls = [] - let toolErrors = [] - if (hasTools) { - const parsed = parseToolCallsFromText(fullContent, { allowedToolNames }) - const nativeCalls = nativeToolAccumulator?.hasAny() ? nativeToolAccumulator.finalize() : [] - assistantContent = parsed.cleanedText - toolCalls = [...nativeCalls, ...parsed.toolCalls].map((call, index) => ({ ...call, index })) - toolErrors = [ - ...parsed.errors, - ...(nativeToolAccumulator?.getErrors() || []) - ] - } - // required 未调用,或只有思考没有可见输出时,共用一次补偿重试。 - const needsRequiredRetry = hasTools && toolCalls.length === 0 && requiresToolCall(toolChoice) - const needsEmptyOutputRetry = toolCalls.length === 0 && toolErrors.length === 0 && !assistantContent.trim() && + // 空回合补偿:只有思考、没有正文时重试一次。 + const needsEmptyOutputRetry = !assistantContent.trim() && !['length', 'max_tokens', 'content_filter', 'refusal'].includes(upstreamFinishReason) - const needsMissingToolRetry = hasTools && toolCalls.length === 0 && toolErrors.length === 0 && - looksLikeUnexecutedToolAction(assistantContent) && - !['length', 'max_tokens', 'content_filter', 'refusal'].includes(upstreamFinishReason) - if (needsRequiredRetry || needsEmptyOutputRetry || needsMissingToolRetry) { - const retryHint = needsRequiredRetry - ? buildRequiredRetryHint(toolChoice) - : (needsMissingToolRetry ? buildMissingToolRetryHint() : buildEmptyOutputRetryHint()) - const retryBody = appendRetryHintToRequestBody(requestBody, retryHint) - logger.warn( - needsRequiredRetry - ? 'tool_choice=required 首次未触发工具调用,进行一次重试' - : (needsMissingToolRetry - ? 'Agent 首次响应只描述了动作但未调用工具,进行一次补偿重试' - : 'Agent 首次响应没有正文或工具调用,进行一次补偿重试'), - 'CHAT' - ) + if (needsEmptyOutputRetry) { + const retryBody = appendRetryHintToRequestBody(requestBody, buildEmptyOutputRetryHint()) + logger.warn('Agent 首次响应没有正文或工具调用,进行一次补偿重试', 'CHAT') try { const retryResp = await requestSender(retryBody, options.upstreamOptions || {}) if (retryResp.status && retryResp.response) { const before = fullContent - nativeToolAccumulator = createNativeToolCallAccumulator({ allowedToolNames }) upstreamFinishReason = null await accumulateUpstream(retryResp.response) if (!upstreamCompleted && !upstreamFinishReason) { + // 消息维持原样: 这条 rama solo se alcanza desde el reintento por + // respuesta vacia, pero el texto es el que ya viajaba al cliente y + // este cambio no altera lo observable. return res.status(502).json({ error: { message: '工具调用重试流在结束标记前断开', @@ -1343,18 +1105,7 @@ const handleNonStreamResponse = async (res, response, enable_thinking, enable_we } }) } - const retriedText = fullContent.slice(before.length) - const parsedRetry = parseToolCallsFromText(retriedText, { allowedToolNames }) - const nativeRetryCalls = nativeToolAccumulator.hasAny() - ? nativeToolAccumulator.finalize() - : [] - toolCalls = [...nativeRetryCalls, ...parsedRetry.toolCalls] - .map((call, index) => ({ ...call, index })) - assistantContent = parsedRetry.cleanedText - toolErrors = [ - ...parsedRetry.errors, - ...nativeToolAccumulator.getErrors() - ] + assistantContent = fullContent.slice(before.length) } } catch (e) { logger.error('Agent 补偿重试失败', 'CHAT', '', e) @@ -1362,18 +1113,7 @@ const handleNonStreamResponse = async (res, response, enable_thinking, enable_we } } - if (hasTools && toolCalls.length === 0 && (toolErrors.length > 0 || requiresToolCall(toolChoice))) { - return res.status(502).json({ - error: { - message: '上游返回了残缺、非法或不存在的工具调用', - type: 'invalid_tool_call', - code: 'invalid_tool_call', - details: toolErrors - } - }) - } - - if (toolCalls.length === 0 && !assistantContent.trim() && + if (!assistantContent.trim() && !['length', 'max_tokens', 'content_filter', 'refusal'].includes(upstreamFinishReason)) { return res.status(502).json({ error: { @@ -1384,11 +1124,7 @@ const handleNonStreamResponse = async (res, response, enable_thinking, enable_we }) } - const finishReason = normalizeOpenAIFinishReason( - upstreamFinishReason, - toolCalls.length > 0, - upstreamCompleted - ) + const finishReason = normalizeOpenAIFinishReason(upstreamFinishReason, upstreamCompleted) if (!finishReason) { return res.status(502).json({ error: { @@ -1419,9 +1155,6 @@ const handleNonStreamResponse = async (res, response, enable_thinking, enable_we if (fullReasoning) { assistantMessage.reasoning_content = fullReasoning } - if (toolCalls.length > 0) { - assistantMessage.tool_calls = toolCalls - } const bodyTemplate = { "id": `chatcmpl-${generateUUID()}`, @@ -1477,11 +1210,13 @@ const handleChatCompletion = async (req, res) => { const response_data = await sendChatRequest(req.body, upstreamOptions) if (!response_data.status || !response_data.response) { - res.status(500) - .json({ - error: response_data.message || "Request failed" - }) - return + // El fallo dice por que fallo: cuota 429 `insufficient_quota`, sobrecarga o + // transporte 503, sin clasificar 502. Antes: un 500 mudo para todo, con el que + // un cliente agentico no podia distinguir "vuelve luego" de "servidor roto". + return writeOpenAIHttpError(res, openAIErrorShape( + response_data.failure || unclassifiedFailure(502), + response_data.message || 'Upstream did not produce a usable response' + )) } // Aviso al cliente cuando el contexto se recortó en silencio. El fallback por fallo @@ -1538,13 +1273,11 @@ const handleChatCompletion = async (req, res) => { // anthropic.js). Cualquier otra cosa conserva el 500 de siempre. const failure = describeUpstreamFailure(error, 500, 503) if (failure.overloaded) { - return writeOpenAIHttpError(res, { - status: failure.status, - message: error.publicMessage || 'Upstream context attachment unavailable; retry', - type: 'server_error', - code: 'upstream_unavailable', - retry_after: failure.retryAfter - }) + // Misma forma que produce el traductor compartido para "upstream sobrecargado". + return writeOpenAIHttpError(res, openAIErrorShape( + failure, + error.publicMessage || 'Upstream context attachment unavailable; retry' + )) } res.status(500) .json({ diff --git a/src/utils/agent-turn-gate.js b/src/utils/agent-turn-gate.js new file mode 100644 index 00000000..e81265dc --- /dev/null +++ b/src/utils/agent-turn-gate.js @@ -0,0 +1,458 @@ +'use strict' + +/** + * La puerta del turno agéntico: qué intento del upstream se acepta y cuál se reintenta. + * + * Entra el snapshot de **un intento** más la política de la superficie; sale el veredicto. + * Acá viven el vocabulario de razones, los dos presupuestos, el mapa de hints de reintento + * con sus constructores y el helper de append. El loop conserva lo que es suyo: el + * presupuesto de intentos, la bandera mutable del cupo de recuperación de protocolo y la + * maquinaria de entrega. La puerta no sabe de presupuestos agotados ni de códigos de cable: + * "intentos agotados" es propiedad del loop, y los códigos terminales son vocabulario de + * cada superficie (ADR 0001). + * + * No lee configuración, no toca red y no loguea: `gate` es una función pura de sus dos + * argumentos, y por eso la tabla de verdad del seam es barata. Hoja en dependencias tampoco + * lo es del todo — importa `tool-prompt.js`, que arrastra el logger, que lee entorno al + * importarse; la pureza que importa es la de `gate`, no la del grafo de imports. + */ + +const { + TOOL_CALL_OPEN, + TOOL_CALL_CLOSE, + buildAgentRetryHint +} = require('./agent-turn.js') +// El heurístico de "prosa que narra una acción" vive con el resto de la maquinaria de +// marcadores (tool-prompt.js): la puerta lo consume, no lo reimplementa. +const { looksLikeUnexecutedToolAction } = require('./tool-prompt.js') + +/** Piso del presupuesto: 1 = una sola generación, sin reintentos. */ +const MIN_ATTEMPT_BUDGET = 1 + +/** Techo del presupuesto: el mismo tope que ya aplicaba el runtime OpenAI. */ +const MAX_ATTEMPT_BUDGET = 6 + +/** Terminaciones que el upstream ya explicó: reintentar no las arregla. */ +const TERMINAL_FINISH_REASONS = new Set(['length', 'max_tokens', 'content_filter', 'refusal']) + +/** Valores de `finishReason` con los que la puerta acepta un turno. */ +const FINISH_TOOL_CALLS = 'tool_calls' +const FINISH_STOP = 'stop' + +/** + * Un vocabulario, un significado por token. El par sinónimo (`required` / `required_tool`) + * se fusiona acá en `required_tool`. `invalid_tool_call` de la superficie OpenAI no existe + * como token: se parte en `tool_error` (errores de herramienta) más `prose_with_tools` + * (prosa junto a llamadas) — fusionarlos enteros ensancharía una regla en silencio. + * + * Los tokens de una sola superficie siguen siendo de esa superficie: `bare`, + * `invalid_control` y `prose_with_tools` los emite la superficie OpenAI; `thought_tool_call` + * y `missing_tool` son de las Anthropic. Los compartidos son `empty`, `required_tool`, + * `tool_error`, `intercepted` y `malformed_protocol`. + */ +const REASONS = Object.freeze({ + EMPTY: 'empty', + BARE: 'bare', + INVALID_CONTROL: 'invalid_control', + REQUIRED_TOOL: 'required_tool', + TOOL_ERROR: 'tool_error', + PROSE_WITH_TOOLS: 'prose_with_tools', + INTERCEPTED: 'intercepted', + MALFORMED_PROTOCOL: 'malformed_protocol', + THOUGHT_TOOL_CALL: 'thought_tool_call', + MISSING_TOOL: 'missing_tool' +}) + +/** + * Las razones que gastan el cupo único de recuperación de protocolo de una petición. El loop + * conserva la bandera mutable; acá vive la regla. Es la unión de las tres disyunciones a mano + * que esto reemplaza — `anthropic.js` (loop streaming y no-stream) listaba + * `intercepted || malformed_protocol || thought_tool_call`; `openai-agent-runtime.js` listaba + * `intercepted || malformed_protocol` (esa superficie no detecta think leak). + */ +const PROTOCOL_RECOVERY_REASONS = new Set([ + REASONS.INTERCEPTED, + REASONS.MALFORMED_PROTOCOL, + REASONS.THOUGHT_TOOL_CALL +]) + +/** + * Un solo significado de "max attempts": total de generaciones del upstream para una + * petición de cliente, **contando la primera**. El runtime OpenAI aplicaba un piso de 2 y + * convertía en 2 el 1 que le pidieran, mientras las superficies Anthropic aplicaban piso 1: + * la misma cifra significaba dos cosas según quién la leyera. + * + * Alcance exacto de lo que esto arregla, para que nadie lo lea de más: el piso de 1 es + * alcanzable **sólo por el valor por petición**. La configuración sigue clampeada a [2, 6] + * en config/index.js — a propósito, y por eso `AGENT_TURN_MAX_ATTEMPTS=1` sigue dando 2. + * Hoy el único llamador que trae un valor por petición son los tests; cuando exista uno de + * producción, el piso de acá ya lo cubre. + * @param {number|string|null} [requested] - valor por petición; null, ausente, no finito o + * ≤ 0 cuentan como "no hay" y cae al de configuración. Ojo: un negativo antes se aplanaba + * a 2 y un `Infinity` a 6 — entradas que ningún llamador real produce, pero que cambian + * de resultado, así que van declaradas y no escondidas. + * @param {number|string} [fallback] - valor de configuración para cuando no lo trae + * @returns {number} intentos totales, entre MIN_ATTEMPT_BUDGET y MAX_ATTEMPT_BUDGET + */ +const resolveAttemptBudget = (requested, fallback) => { + const wanted = Number(requested) + const base = Number.isFinite(wanted) && wanted > 0 ? wanted : Number(fallback) + if (!Number.isFinite(base) || base <= 0) return MIN_ATTEMPT_BUDGET + return Math.min(MAX_ATTEMPT_BUDGET, Math.max(MIN_ATTEMPT_BUDGET, base)) +} + +// --------------------------------------------------------------------------- +// Hints de reintento: un mapa indexado por el vocabulario, un texto por token +// --------------------------------------------------------------------------- + +/** + * 构建 required 重试提示 + * @param {string|Object} toolChoice - 内部 tool_choice + * @returns {string} 提示文本 + */ +const buildRequiredToolRetryHint = (toolChoice) => { + if (toolChoice && typeof toolChoice === 'object' && toolChoice.function?.name) { + return `You did not call any tool. You MUST now call \`${toolChoice.function.name}\` using the ${TOOL_CALL_OPEN}...${TOOL_CALL_CLOSE} format.` + } + return `You did not call any tool. You MUST now call exactly one tool using the ${TOOL_CALL_OPEN}...${TOOL_CALL_CLOSE} format.` +} + +/** + * El hint del caso vacío. Antes vivía en tres módulos (`anthropic.js`, `chat.js` y el + * runtime OpenAI arman el suyo con `buildAgentRetryHint('empty')`); este es el texto que + * el corpus grabó para las superficies Anthropic, byte a byte. + */ +const buildEmptyOutputRetryHint = () => [ + 'Your previous reply produced no visible final answer or executable tool call.', + `Continue the Agent task now. If any action remains, emit the required \`${TOOL_CALL_OPEN}\` block immediately with no preamble.`, + 'Only give a normal final answer when the task is actually complete; do not repeat hidden reasoning.' +].join(' ') + +/** El hint de "describió la acción y no la ejecutó" (`missing_tool`), solo Anthropic. */ +const buildMissingToolRetryHint = () => [ + 'Your previous reply described an action but did not execute any tool call.', + `Perform that action now by emitting the real \`${TOOL_CALL_OPEN}\` block immediately with no preamble.`, + 'Do not describe the action again or claim completion without a tool result.' +].join(' ') + +/** + * 工具错误的重试提示。基础文本复用 agent-turn.js 的通用提示;当错误是编造的工具名时, + * 补上真实的名字 —— 那是让这类错误可恢复的唯一信息。原生调用的参数不合法 + * (invalid_arguments / schema_mismatch)时,点名该工具:模型要重发的是参数,不是名字。 + * @param {Array} errors - 本轮的工具错误 + * @param {Array} allowedToolNames - 本次请求真正提供的工具名 + * @returns {string} 提示文本 + */ +const buildToolErrorRetryHint = (errors, allowedToolNames) => { + const base = buildAgentRetryHint('invalid_tool_call') + const unknown = [...new Set( + errors.filter(e => e?.type === 'unknown_tool').map(e => e.name).filter(Boolean) + )] + const badArguments = [...new Set( + errors.filter(e => e?.type === 'invalid_arguments' || e?.type === 'schema_mismatch').map(e => e.name).filter(Boolean) + )] + const lines = [base] + if (unknown.length && allowedToolNames?.length) { + lines.push( + `The tool name(s) ${unknown.join(', ')} do not exist.`, + `Use ONLY these exact tool names: ${allowedToolNames.join(', ')}.` + ) + } + if (badArguments.length) { + lines.push(`Your arguments for tool ${badArguments.join(', ')} were not a valid JSON object or missed required keys. Re-emit the call with a complete JSON object that matches the tool's input schema.`) + } + return lines.join('\n') +} + +/** + * El mapa: un token del vocabulario, un constructor de hint. Los cuerpos compartidos salen + * de `agent-turn.js#buildAgentRetryHint` (una sola copia de esos textos); los tres tokens + * que ese mapa no tenía (`required_tool`, `missing_tool`, `tool_error`) entran con el texto + * exacto que ya producían los constructores locales del controlador. + * + * `prose_with_tools` es el detalle con el que la superficie OpenAI rechaza prosa junto a + * llamadas: su hint hoy es el cuerpo `invalid_tool_call` de agent-turn.js, así que el token + * lo conserva textualmente. La forma de hint de `required_tool` que usa la superficie + * OpenAI (el cuerpo genérico, sin `tool_choice` que nombrar) es otra: la decide el ticket + * que cablea esa superficie, no esta. + */ +const RETRY_HINT_BUILDERS = Object.freeze({ + [REASONS.EMPTY]: () => buildEmptyOutputRetryHint(), + [REASONS.BARE]: () => buildAgentRetryHint('bare'), + [REASONS.INVALID_CONTROL]: () => buildAgentRetryHint('invalid_control'), + [REASONS.REQUIRED_TOOL]: (snapshot, context) => buildRequiredToolRetryHint(context?.toolChoice), + [REASONS.TOOL_ERROR]: (snapshot, context) => buildToolErrorRetryHint(snapshot.toolErrors || [], context?.allowedToolNames), + [REASONS.PROSE_WITH_TOOLS]: () => buildAgentRetryHint('invalid_tool_call'), + [REASONS.INTERCEPTED]: () => buildAgentRetryHint('intercepted'), + [REASONS.MALFORMED_PROTOCOL]: () => buildAgentRetryHint('malformed_protocol'), + [REASONS.THOUGHT_TOOL_CALL]: () => buildAgentRetryHint('thought_tool_call'), + [REASONS.MISSING_TOOL]: () => buildMissingToolRetryHint() +}) + +/** + * El hint de una razón, con los dos apéndices que hoy agrega el controlador: + * - `required_tool` / `missing_tool` ganan la prioridad sobre `intercepted` y lo esconden; + * el hint no cambia la prioridad — solo lleva el hecho clave: la llamada no llegó. + * - `required_tool` / `tool_error` tapan `thought_tool_call`; el hint lleva igual el hecho + * de que la llamada quedó escrita en la razón oculta del modelo. + * @param {string} reason - token del vocabulario + * @param {Object} snapshot - el snapshot del intento (de acá salen errores y evidencia) + * @param {{ toolChoice?: string|Object, allowedToolNames?: Array }} [context] + * @returns {string} hint de reintento + */ +const retryHintFor = (reason, snapshot, context = {}) => { + const build = RETRY_HINT_BUILDERS[reason] + let hint = build ? build(snapshot || {}, context) : buildAgentRetryHint(reason) + if ((reason === REASONS.REQUIRED_TOOL || reason === REASONS.MISSING_TOOL) && + (snapshot?.interceptedToolNames || []).length > 0) { + hint = `${hint}\n${buildAgentRetryHint('intercepted')}` + } + if ((reason === REASONS.REQUIRED_TOOL || reason === REASONS.TOOL_ERROR) && snapshot?.thinkEvidence) { + hint = `${hint}\n${buildAgentRetryHint('thought_tool_call')}` + } + return hint +} + +// --------------------------------------------------------------------------- +// Mensajes de agotamiento: el segundo mapa indexado por el mismo vocabulario +// --------------------------------------------------------------------------- + +/** + * Qué se le dice al cliente cuando el presupuesto de intentos se agota y el último intento + * seguía rechazado. Vivía en `openai-agent-runtime.js` con la clave `invalid_tool_call` —la + * misma condición que este vocabulario parte en `tool_error` (errores de herramienta) y + * `prose_with_tools` (prosa junto a llamadas)—, y las dos mitades conservan el texto de la + * clave vieja: el renombre es del vocabulario, no del mensaje que ya viajaba al cliente. + * + * El mapa NO lleva el status ni el código de cable: estos son vocabulario de cada superficie + * (ADR 0001) y los pone el runtime que consume el mapa. + */ +const EXHAUSTED_TURN_MESSAGES = Object.freeze({ + [REASONS.EMPTY]: '上游连续只返回思考内容,没有给出可执行工具调用或最终答复', + [REASONS.BARE]: '上游连续返回未声明完成状态的文本,已阻止 Agent 将未完成任务误判为结束', + [REASONS.INVALID_CONTROL]: '上游连续返回无效的 Agent 完成标记', + [REASONS.TOOL_ERROR]: '上游连续返回残缺、非法或不存在的工具调用', + [REASONS.PROSE_WITH_TOOLS]: '上游连续返回残缺、非法或不存在的工具调用', + [REASONS.REQUIRED_TOOL]: '上游连续违反 tool_choice,未返回要求的工具调用', + [REASONS.INTERCEPTED]: '上游的工具调用被平台拦截,重试后仍未恢复', + [REASONS.MALFORMED_PROTOCOL]: '上游持续返回残缺的工具调用协议,未能恢复为可执行调用' +}) + +/** + * Añade un hint de reintento al último mensaje del cuerpo interno. Una sola implementación + * para las tres superficies: `header` es lo único que cambia entre ellas (el Anthropic manda + * `# Tool-call retry` delante; el runtime OpenAI no manda encabezado) y va donde hoy lo + * mandan sus llamadores — en los dos brazos que **agregan a un texto existente**. Los brazos + * que insertan un bloque nuevo lo hacen sin encabezado, como hoy. + * + * Devuelve un clon: el cuerpo original se reusa entre reintentos y no puede llevar la marca + * del intento anterior. + * @param {Object} body - cuerpo interno de la petición + * @param {string} hint - hint de reintento + * @param {{ header?: string|null }} [options] - encabezado a intercalar, si la superficie lo usa + * @returns {Object} cuerpo nuevo con el hint al final + */ +const appendRetryHint = (body, hint, { header = null } = {}) => { + const clone = body && typeof body === 'object' + ? JSON.parse(JSON.stringify(body)) + : {} + const messages = Array.isArray(clone.messages) ? clone.messages : [] + const separator = header ? `\n\n${header}\n` : '\n\n' + if (messages.length === 0) { + messages.push({ role: 'user', content: hint }) + } else { + const last = messages[messages.length - 1] + if (typeof last.content === 'string') { + last.content = `${last.content}${separator}${hint}` + } else if (Array.isArray(last.content)) { + const textPart = last.content.find(part => part?.type === 'text') + if (textPart) textPart.text = `${textPart.text || ''}${separator}${hint}` + else last.content.unshift({ type: 'text', text: hint }) + } else { + last.content = hint + } + } + clone.messages = messages + return clone +} + +// --------------------------------------------------------------------------- +// La puerta +// --------------------------------------------------------------------------- + +/** + * Lo que la superficie OpenAI hoy retira de la entrega cuando acepta una ronda por sus + * llamadas nativas: el texto que la acompaña puede traer llamadas de texto mal escritas o + * residuo de protocolo, y ese texto no viaja. + */ +const suppressForNativeAccept = (snapshot) => + (snapshot.textToolErrors || []).length > 0 || snapshot.orphanResidue === true + +/** + * La misma idea en la rama de la ronda cortada: acá el veto mira **todos** los errores (no + * solo los del canal de texto) y, además, la prosa que la política no permite junto a tools. + */ +const suppressForCutAccept = (snapshot, policy) => + (snapshot.toolErrors || []).length > 0 || + snapshot.orphanResidue === true || + (policy?.proseWithTools !== true && String(snapshot.visibleText || '').trim() !== '') + +const accept = (finishReason, suppressVisibleText = false) => ({ + verdict: 'accept', + finishReason, + reason: null, + suppressVisibleText +}) + +const retry = (reason) => ({ + verdict: 'retry', + finishReason: null, + reason, + suppressVisibleText: false +}) + +/** + * La decisión de un intento. Función pura de un snapshot más una política. + * + * Snapshot — describe **exactamente un intento** (lo compartido entre intentos, el cupo de + * recuperación de protocolo y si alguna vez se entregó texto, se queda en el loop): + * finishReason razón cruda del upstream + * visibleText texto visible **de este intento** (el insumo de detección) + * controlKind 'final' | 'blocked' | 'empty' | 'invalid_control' | 'bare' | null + * (null = una superficie sin vocabulario de control, como las Anthropic) + * toolCalls llamadas admisibles de este intento + * toolErrors todos los errores de herramienta del intento + * textToolErrors el subconjunto del canal de texto (decide con finish terminal) + * nativeToolCalls llamadas estructuradas admitidas por el gate de schema + * interceptedToolNames frames role:function que la plataforma se comió + * thinkEvidence una llamada quedó escrita en la razón oculta + * callsDelivered un bloque tool_use ya llegó al cliente (no se puede retractar) + * textChannelCut la guarda de fuga cortó el texto del intento a mitad de stream + * orphanResidue residuo de protocolo malformado en visibleText + * hasTools la petición declaró herramientas + * requiresToolCall el tool_choice de la petición exige una llamada + * Los dos últimos son los únicos hechos de **petición** que la puerta necesita; hoy viven en + * `options` del runtime OpenAI y en el `ctx` de los controladores Anthropic. + * + * Política — cuatro campos con nombre, decididos por el llamador (la puerta no lee el + * singleton de configuración): + * proseWithTools ¿la prosa junto a llamadas se acepta? (OpenAI no, Anthropic sí) + * acceptBareFinal ¿la prosa sin envoltorio de cierre es una respuesta final? + * toolErrorsBeforeRequired ¿los errores de herramienta vetan antes que `required_tool`? + * toolErrorsVetoWithCalls ¿un error de herramienta veta aunque haya una llamada parseada? + * + * @param {Object} snapshot - un intento + * @param {Object} policy - la política de la superficie + * @returns {{ verdict: 'accept'|'retry', finishReason: string|null, reason: string|null, suppressVisibleText: boolean }} + */ +const gate = (snapshot, policy) => { + const s = snapshot || {} + const p = policy || {} + const terminal = TERMINAL_FINISH_REASONS.has(s.finishReason) + const hasTools = s.hasTools !== false + const visibleText = typeof s.visibleText === 'string' ? s.visibleText : '' + const calls = s.toolCalls || [] + const nativeToolCalls = s.nativeToolCalls || [] + const intercepted = s.interceptedToolNames || [] + const controlKind = s.controlKind || null + // Con finish terminal, el error nativo (el snapshot truncado del acumulador) no veta: la + // fuente de texto mantiene su comportamiento. Sin terminal, vetan los dos canales. + const toolErrors = terminal ? (s.textToolErrors || []) : (s.toolErrors || []) + + // El cliente ya tiene un bloque tool_use: no hay nada que retractar. Este es el campo que + // hace que stream y no-stream sean la misma decisión. + if (s.callsDelivered === true) return accept(FINISH_TOOL_CALLS) + + // Llamadas nativas estructuradas: evidencia más dura que cualquier veto textual, va antes + // de los errores y de "la prosa no convive con tools". + if (nativeToolCalls.length > 0) { + return accept(FINISH_TOOL_CALLS, suppressForNativeAccept(s)) + } + + // Ronda cortada por la guarda de fuga con llamadas admitidas: se entrega siempre (rechazarla + // reintenta la fuga recién detenida, y el reintento cae en el chat_id cuya generación + // abortada sigue viva en Qwen → CHAT_IN_PROGRESS). + if (s.textChannelCut === true && calls.length > 0) { + return accept(FINISH_TOOL_CALLS, suppressForCutAccept(s, p)) + } + + // Errores de herramienta: la precedencia respecto de `required_tool` — y si vetar o no + // cuando además hay una llamada parseada — es política de la superficie. + const toolErrorVeto = toolErrors.length > 0 && + (calls.length === 0 || p.toolErrorsVetoWithCalls === true) + if (p.toolErrorsBeforeRequired === true && toolErrorVeto) return retry(REASONS.TOOL_ERROR) + + // tool_choice exigía una llamada y este intento no entregó ninguna. + if (hasTools && s.requiresToolCall === true && calls.length === 0) return retry(REASONS.REQUIRED_TOOL) + + if (toolErrorVeto) return retry(REASONS.TOOL_ERROR) + + // Terminaciones que el upstream ya explicó: la ronda se entrega como está, reintentar no + // las arregla. Va después de los vetos —con finish terminal el error del canal de texto + // todavía veta, y `required_tool` también— y antes de la rama de llamadas, para que una + // ronda truncada con llamadas se entregue en vez de reintentarse por su prosa. + if (terminal) return accept(FINISH_STOP) + + // Llamadas admisibles: la superficie OpenAI veta la prosa que las acompaña; las Anthropic + // aceptan — el cliente recibe bloques tool_use discretos y puede actuar con lo que llegó. + if (calls.length > 0) { + if (p.proseWithTools !== true && (controlKind !== 'empty' || visibleText.trim())) { + return retry(REASONS.PROSE_WITH_TOOLS) + } + return accept(FINISH_TOOL_CALLS) + } + + // Evidencia de protocolo. Con finish terminal no se reintenta (la generación ya terminó + // por una razón que el reintento no arregla) y sin herramientas no hay evidencia que valga. + if (hasTools && !terminal) { + // intercepted primero: el frame descartado es la evidencia más fuerte. + if (intercepted.length > 0) return retry(REASONS.INTERCEPTED) + // El cupo de recuperación de protocolo ya gastado no reintenta por residuo — pero el + // residuo sigue contando para la supresión de la entrega (ver `suppressForNativeAccept`). + // Sólo la superficie que lleva ese cupo lo declara; donde el campo no llega, la regla es + // la de siempre. + if (s.orphanResidue === true && s.protocolRecoverySpent !== true) return retry(REASONS.MALFORMED_PROTOCOL) + if (s.thinkEvidence === true) return retry(REASONS.THOUGHT_TOOL_CALL) + // Solo en una superficie sin vocabulario de control: donde hay envoltorio, la prosa la + // deciden las reglas del envoltorio (la superficie OpenAI no detecta `missing_tool`). + if (controlKind === null && looksLikeUnexecutedToolAction(visibleText)) { + return retry(REASONS.MISSING_TOOL) + } + } + + // El envoltorio de cierre (vocabulario de la superficie OpenAI; las Anthropic pelan los + // tags sin interpretarlos, así que llegan acá con controlKind null). + if (controlKind === 'final' || controlKind === 'blocked') { + return visibleText.trim() ? accept(FINISH_STOP) : retry(REASONS.EMPTY) + } + if (controlKind === 'empty') return retry(REASONS.EMPTY) + if (controlKind === 'invalid_control') return retry(REASONS.INVALID_CONTROL) + if (controlKind === 'bare') { + return p.acceptBareFinal === true ? accept(FINISH_STOP) : retry(REASONS.BARE) + } + + // Sin envoltorio: `empty` es alcance de intento (el texto que cuenta es el de este intento, + // no el acumulado), y la prosa que queda es una respuesta final si la superficie lo dice. + if (!terminal && !visibleText.trim()) return retry(REASONS.EMPTY) + return p.acceptBareFinal === true ? accept(FINISH_STOP) : retry(REASONS.BARE) +} + +module.exports = { + gate, + REASONS, + PROTOCOL_RECOVERY_REASONS, + TERMINAL_FINISH_REASONS, + FINISH_STOP, + FINISH_TOOL_CALLS, + resolveAttemptBudget, + RETRY_HINT_BUILDERS, + EXHAUSTED_TURN_MESSAGES, + retryHintFor, + appendRetryHint, + buildRequiredToolRetryHint, + buildEmptyOutputRetryHint, + buildMissingToolRetryHint, + buildToolErrorRetryHint, + MIN_ATTEMPT_BUDGET, + MAX_ATTEMPT_BUDGET +} diff --git a/src/utils/openai-agent-runtime.js b/src/utils/openai-agent-runtime.js index 833d66f1..b5e59344 100644 --- a/src/utils/openai-agent-runtime.js +++ b/src/utils/openai-agent-runtime.js @@ -9,7 +9,10 @@ const { const { consumeSSEStream, createUpstreamResponseFilter } = require('./sse.js') const { mergeUpstreamUsage } = require('./precise-tokenizer.js') const { createUpstreamDeltaNormalizer, createClientToolNamePredicate } = require('./chat-helpers.js') -const { assertNoUpstreamFailure, UpstreamResponseError, isRateLimitError, isWafChallengeError } = require('./upstream-error.js') +const { + assertNoUpstreamFailure, UpstreamResponseError, isRateLimitError, isWafChallengeError, + openAIErrorShape, unclassifiedFailure +} = require('./upstream-error.js') const { recordFailedAccount, createAccountReplayBody } = require('./agent-account-failover.js') const { parseAgentControlText, @@ -23,15 +26,18 @@ const { createTextChannelRunawayGuard } = require('./agent-turn.js') const config = require('../config/index.js') +const { + gate, + REASONS, + TERMINAL_FINISH_REASONS, + PROTOCOL_RECOVERY_REASONS, + EXHAUSTED_TURN_MESSAGES, + retryHintFor, + appendRetryHint, + resolveAttemptBudget +} = require('./agent-turn-gate.js') const { logger } = require('./logger.js') -const NON_RETRYABLE_FINISH_REASONS = new Set([ - 'length', - 'max_tokens', - 'content_filter', - 'refusal' -]) - /** * Rebasa los spans de residuo de coordenadas de `cleanedText` a las de `visibleText`. * @@ -255,8 +261,8 @@ const collectOpenAIAgentAttempt = async (upstreamResponse, options = {}) => { cleanedText: streamedRawText, toolCalls: streamedCalls, // Una ronda cortada NO superficializa errores del parser — ni los del push disparador ni - // los de pushes anteriores. evaluateOpenAIAgentAttempt mira `toolErrors.length > 0` ANTES - // que `toolCalls.length > 0`: cualquier error superviviente reintentaria la ronda y + // los de pushes anteriores. La puerta mira `toolErrors` ANTES que `toolCalls`: cualquier + // error superviviente reintentaria la ronda y // volveria a lanzar la fuga que el corte acaba de detener, hasta agotar intentos y morir // en 502. Es la paridad con anthropic.js, donde decideRetryReason corta en seco con // `if (emittedCalls) return null` (:1080) / `if (toolCalls.length > 0) return null` @@ -459,7 +465,8 @@ const collectOpenAIAgentAttempt = async (upstreamResponse, options = {}) => { const textChannelCut = !!textRunaway?.cutRule() // Tras un corte no se hace flush de ningun parser de texto: lo que queda en el buffer es // el resto del push descontrolado (medio trigger / medio payload) y el flush lo condenaria - // como truncated_tool_call -> toolErrors>0 -> reintento (evaluate :461), anulando el corte. + // como truncated_tool_call -> toolErrors>0 -> reintento (la puerta veta antes de la rama de + // llamadas), anulando el corte. if (reasoningStreamParser && !textChannelCut) { const streamed = reasoningStreamParser.flush() await emitReasoningDelta(streamed.textDelta) @@ -601,122 +608,121 @@ const requiresToolCall = (toolChoice) => { return !!(toolChoice && typeof toolChoice === 'object' && toolChoice.type === 'function' && toolChoice.function?.name) } -const evaluateOpenAIAgentAttempt = (attempt, options = {}) => { - const finishReason = attempt.upstreamFinishReason - if (NON_RETRYABLE_FINISH_REASONS.has(finishReason)) { - const normalized = finishReason === 'max_tokens' ? 'length' : finishReason - return { accepted: true, finishReason: normalized, retryReason: null } - } - // 过闸的原生调用是结构化帧,比文本启发式更强的证据:有一个就接纳本轮 —— 排在 - // toolErrors 否决与"正文不得与工具并存"之前,不翻 agentTurnAllowProseWithTools。 - // 调用前的干净正文随 visibleText 交付;调用后的叙述在采集时就已丢弃。但被跳过的两道 - // 否决恰恰说明 visibleText 里可能混着写坏的文本 [TOOL CALL](文本来源的解析错误 / - // 孤儿协议残渣):这种正文不交付 —— suppressVisibleText 让交付层把 content 置空, - // tool_calls 照常。 - if ((attempt.nativeToolCalls?.length || 0) > 0) { - const suppressVisibleText = (attempt.textToolErrors?.length || 0) > 0 || - containsOrphanProtocolResidue(attempt.visibleText) - return { accepted: true, finishReason: 'tool_calls', retryReason: null, suppressVisibleText } - } - // Ronda cortada por la guarda de fuga con llamadas admitidas: SIEMPRE se entrega (paridad - // con anthropic.js decideRetryReason :1080/:1749 y la promesa de settledTextRound). Va ANTES - // del veto por toolErrors y de "prosa no coexiste con tools": rechazarla reintenta la fuga - // recién detenida y, peor, el reintento cae en el mismo chat_id cuya generación abortada - // sigue viva en Qwen → CHAT_IN_PROGRESS → 502 (incidente qwen-next 2026-09-06 20:29, gate - // estricto: narración previa a la llamada + corte por duplicado). La prosa previa al corte - // viaja sólo si la config la permite; con gate estricto se suprime en vez de rechazar. - if (attempt.textChannelCut === true && attempt.toolCalls.length > 0) { - const suppressVisibleText = (attempt.toolErrors?.length || 0) > 0 || - containsOrphanProtocolResidue(attempt.visibleText) || - (!config.agentTurnAllowProseWithTools && !!attempt.visibleText.trim()) - return { accepted: true, finishReason: 'tool_calls', retryReason: null, suppressVisibleText } - } - if (attempt.toolErrors.length > 0) { - return { accepted: false, finishReason: null, retryReason: 'invalid_tool_call', detail: 'tool_errors' } - } - if (attempt.toolCalls.length > 0) { - if (!config.agentTurnAllowProseWithTools && - (attempt.controlKind !== 'empty' || attempt.visibleText.trim())) { - return { accepted: false, finishReason: null, retryReason: 'invalid_tool_call', detail: 'prose_with_tools' } - } - return { accepted: true, finishReason: 'tool_calls', retryReason: null } - } - if (requiresToolCall(options.tool_choice)) { - return { accepted: false, finishReason: null, retryReason: 'required_tool' } - } - // 协议恢复防御(与 Anthropic 两个循环同族)。必须排在 final/blocked 接纳之前: - // 事故正是以 包着的失败叙述被当成合法完结交付出去的。 - // - intercepted:role:function 丢弃帧 = 平台吃掉了模型的原生调用,只剩叙述。 - // - malformed_protocol:方括号协议写坏(孤儿闭标记 / 开头裸负载)整段泄漏为 - // 可见正文。只是重试信号,泄漏的 JSON 永远不执行。 - // intercepted 在前——丢弃帧是更强的证据。protocol_recovery_used 表示共享的 - // 一次性恢复名额已用:跳过两个检查,让回合按原有规则交付(原样交付胜过死循环)。 - if (options.has_tools !== false && !options.protocol_recovery_used) { - if ((attempt.interceptedToolNames?.length || 0) > 0) { - return { accepted: false, finishReason: null, retryReason: 'intercepted' } - } - if (containsOrphanProtocolResidue(attempt.visibleText)) { - return { accepted: false, finishReason: null, retryReason: 'malformed_protocol' } - } - } - if (attempt.controlKind === 'final' || attempt.controlKind === 'blocked') { - if (attempt.visibleText.trim()) { - return { accepted: true, finishReason: 'stop', retryReason: null } - } - return { accepted: false, finishReason: null, retryReason: 'empty' } - } - if (attempt.controlKind === 'empty') { - return { accepted: false, finishReason: null, retryReason: 'empty' } - } - if (attempt.controlKind === 'invalid_control') { - return { accepted: false, finishReason: null, retryReason: 'invalid_control' } - } - if (config.agentTurnAcceptBareFinal && attempt.visibleText.trim()) { - return { accepted: true, finishReason: 'stop', retryReason: null } - } - return { accepted: false, finishReason: null, retryReason: 'bare' } -} +/** + * La política de esta superficie, en los cuatro campos nombrados que la puerta entiende. + * Hasta el ticket 07 estas cuatro reglas vivían como ramas propias dentro de una evaluación + * local; los valores son los de siempre — este cableado no cambia ninguno. + */ +const openAIAgentGatePolicy = () => ({ + // 正文不得与工具并存: con agentTurnAllowProseWithTools=false la prosa junto a llamadas se + // rechaza (asimetría deliberada con las superficies Anthropic, que sí la aceptan). + proseWithTools: config.agentTurnAllowProseWithTools === true, + // 裸正文不是完成: con agentTurnAcceptBareFinal=false la prosa sin envoltorio se reintenta. + acceptBareFinal: config.agentTurnAcceptBareFinal === true, + // El veto por error de herramienta manda sobre `required_tool`: el veto miraba ANTES de + // `requiresToolCall` (una llamada inventada no satisface el tool_choice). + toolErrorsBeforeRequired: true, + // Y veta aunque haya una llamada parseada: una llamada parcial es una acción + // silenciosamente equivocada (paridad con anthropic.js, donde decideRetryReason entrega + // la llamada buena — allá el cliente recibe bloques discretos y puede actuar con lo que llegó). + toolErrorsVetoWithCalls: true +}) -const appendRetryHint = (requestBody, hint) => { - const clone = requestBody && typeof requestBody === 'object' - ? JSON.parse(JSON.stringify(requestBody)) - : {} - const messages = Array.isArray(clone.messages) ? clone.messages : [] - if (messages.length === 0) { - messages.push({ role: 'user', content: hint }) - } else { - const last = messages[messages.length - 1] - if (typeof last.content === 'string') { - last.content = `${last.content}\n\n${hint}` - } else if (Array.isArray(last.content)) { - const textPart = last.content.find(part => part?.type === 'text') - if (textPart) textPart.text = `${textPart.text || ''}\n\n${hint}` - else last.content.unshift({ type: 'text', text: hint }) - } else { - last.content = hint - } - } - clone.messages = messages - return clone +/** + * El snapshot de un intento tal como lo ve la puerta: solo hechos de ESTE intento. + * + * `protocolRecoverySpent` no es un hecho del intento sino del loop (la puerta no tiene cupo): + * cuando el cupo compartido ya se gastó, el snapshot se presenta SIN la evidencia que ese + * cupo existe para gastar —intercepted / malformed_protocol— y el intento se juzga por las + * reglas normales. Es la semántica de siempre: 第二次拦截/残缺按原样交付, 原样交付胜过死循环. + * + * `hasTools` va explícito: la puerta asume `true` cuando falta (correcto para las superficies + * Anthropic, una herencia silenciosa aquí). Los dos hechos de petición son los únicos que no + * salen del intento. + */ +const buildOpenAIAgentGateSnapshot = (attempt, options = {}, protocolRecoverySpent = false) => ({ + finishReason: attempt.upstreamFinishReason, + visibleText: attempt.visibleText, + controlKind: attempt.controlKind, + toolCalls: attempt.toolCalls, + toolErrors: attempt.toolErrors, + textToolErrors: attempt.textToolErrors, + nativeToolCalls: attempt.nativeToolCalls, + interceptedToolNames: protocolRecoverySpent ? [] : attempt.interceptedToolNames, + // Esta superficie no detecta el think leak ni la narración sin ejecución: son vocabulario + // de las superficies sin envoltorio de control (parseAgentControlText nunca devuelve null). + thinkEvidence: false, + // Las llamadas de esta superficie solo viajan al aceptar el intento: nunca hay un bloque + // ya entregado al cliente cuando la puerta juzga. + callsDelivered: false, + textChannelCut: attempt.textChannelCut === true, + // El residuo se mide SIEMPRE y en crudo: la supresión de la entrega no depende del cupo de + // recuperación — el código viejo llamaba containsOrphanProtocolResidue sin condición para + // decidir si el texto viajaba. Lo que el cupo gatea es el REINTENTO por malformed_protocol, + // y eso viaja como campo aparte para que la puerta no tenga que adivinar cuál de los dos + // usos está mirando. + orphanResidue: containsOrphanProtocolResidue(attempt.visibleText), + protocolRecoverySpent, + hasTools: options.has_tools !== false, + requiresToolCall: requiresToolCall(options.tool_choice) +}) + +/** + * Las redacciones propias de esta superficie, para los tokens donde el texto del mapa de la + * puerta dice otra cosa: su envoltorio de control tiene su propio hint (`empty`, + * `required_tool`) y `prose_with_tools` conserva el cuerpo `invalid_tool_call` de + * agent-turn.js, que es el texto que la puerta da a ese token. El resto (bare, invalid_control, + * intercepted, malformed_protocol) sale del mapa compartido, byte a byte como siempre. + */ +// Sólo las dos razones cuyo TEXTO difiere del que sirve la puerta: `empty` y `required_tool` +// tienen redacción propia en esta superficie. `prose_with_tools` no está acá porque el +// constructor de la puerta ya devuelve exactamente este texto — tenerlo en los dos lados era +// un segundo hogar que podía derivar sin que nada lo notara. +const OPENAI_LOCAL_HINTS = Object.freeze({ + [REASONS.EMPTY]: () => buildAgentRetryHint('empty'), + [REASONS.REQUIRED_TOOL]: () => buildAgentRetryHint('required_tool') +}) + +/** + * El hint de una razón. `tool_error` entra por el constructor compartido y ESO es el cambio + * declarado del ticket 07: con el vocabulario fusionado el texto gana el apéndice que nombra + * los nombres inventados y los argumentos inválidos — el único texto que hace recuperable un + * nombre desconocido. `allowedToolNames` viaja siempre: omitirlo descarta el apéndice en + * silencio (la trampa anotada por la revisión del ticket 05). + */ +const openAIRetryHint = (reason, snapshot, options = {}) => { + const local = OPENAI_LOCAL_HINTS[reason] + if (local) return local() + return retryHintFor(reason, snapshot, { + toolChoice: options.tool_choice, + allowedToolNames: options.allowed_tool_names + }) } +/** La razón de cierre que viaja al cliente. La puerta acepta con `stop`/`tool_calls`: conoce + * los tokens de truncamiento (los importa para decidir) pero no los emite, así que una ronda + * aceptada por finish terminal conserva el suyo (max_tokens se normaliza a length, como + * siempre). */ +const wireFinishReason = (attempt, evaluation) => + TERMINAL_FINISH_REASONS.has(attempt.upstreamFinishReason) + ? (attempt.upstreamFinishReason === 'max_tokens' ? 'length' : attempt.upstreamFinishReason) + : evaluation.finishReason + +/** + * El error de agotamiento. El texto por razón lo sirve el mapa de la puerta + * (`EXHAUSTED_TURN_MESSAGES`); el status y el código de cable son vocabulario de ESTA + * superficie y se quedan aquí. El código de cable no cambió con el renombre del vocabulario: + * las dos mitades del viejo `invalid_tool_call` (tool_error / prose_with_tools) siguen + * saliendo como `invalid_tool_call`. + */ const exhaustedError = (attempt, retryReason) => { - if (retryReason === 'empty' && !String(attempt?.reasoning || '').trim()) { + if (retryReason === REASONS.EMPTY && !String(attempt?.reasoning || '').trim()) { return { status: 503, message: '上游连续返回空 Agent 回合,任务状态未被标记为完成', code: 'upstream_unavailable' } } - const messages = { - empty: '上游连续只返回思考内容,没有给出可执行工具调用或最终答复', - bare: '上游连续返回未声明完成状态的文本,已阻止 Agent 将未完成任务误判为结束', - invalid_control: '上游连续返回无效的 Agent 完成标记', - invalid_tool_call: '上游连续返回残缺、非法或不存在的工具调用', - required_tool: '上游连续违反 tool_choice,未返回要求的工具调用', - intercepted: '上游的工具调用被平台拦截,重试后仍未恢复', - malformed_protocol: '上游持续返回残缺的工具调用协议,未能恢复为可执行调用' - } return { // 502, no 429. Nada de esto fue un límite de tasa: es un desacuerdo de protocolo con el // upstream. Con 429, chat.js#writeOpenAIHttpError lo etiquetaba `rate_limit_error`, y un @@ -724,8 +730,10 @@ const exhaustedError = (attempt, retryReason) => { // entero" — multiplicando el gasto de cuota de la cuenta contra la que ya se falló. // El 429 real (Qwen RateLimited) sigue saliendo por chat.image.video.js. status: 502, - message: messages[retryReason] || '上游未能生成有效的 Agent 回合', - code: retryReason === 'invalid_tool_call' ? 'invalid_tool_call' : 'upstream_agent_turn_incomplete' + message: EXHAUSTED_TURN_MESSAGES[retryReason] || '上游未能生成有效的 Agent 回合', + code: retryReason === REASONS.TOOL_ERROR || retryReason === REASONS.PROSE_WITH_TOOLS + ? 'invalid_tool_call' + : 'upstream_agent_turn_incomplete' } } @@ -735,10 +743,7 @@ const exhaustedError = (attempt, retryReason) => { */ const runOpenAIAgentTurn = async (initialResponse, options = {}) => { const requestSender = options.sendChatRequest - const maxAttempts = Math.min( - 6, - Math.max(2, Number(options.agent_turn_max_attempts) || config.agentTurnMaxAttempts) - ) + const maxAttempts = resolveAttemptBudget(options.agent_turn_max_attempts, config.agentTurnMaxAttempts) let currentResponse = initialResponse let lastAttempt = null let lastEvaluation = null @@ -768,8 +773,9 @@ const runOpenAIAgentTurn = async (initialResponse, options = {}) => { } } let attemptsMade = 0 - // 协议恢复重试(intercepted / malformed_protocol 共享)整个请求只允许一次。 - // 用过之后 evaluate 会跳过这两个检查,让第二次拦截/残缺按原有规则原样交付。 + // 协议恢复重试(intercepted / malformed_protocol 共享)整个请求只允许一次。用过之后 el + // snapshot se presenta sin esa evidencia, y el segundo interceptado/残缺 se juzga por las + // reglas normales (原样交付胜过死循环). let protocolRecoveryRetried = false const mergePresent = (base, extra) => { @@ -844,15 +850,16 @@ const runOpenAIAgentTurn = async (initialResponse, options = {}) => { upstreamContext = { chatId: retryResponse.chatId || null, parentId: null, responseId: null } continue } - const evaluation = evaluateOpenAIAgentAttempt(attempt, { - ...options, - protocol_recovery_used: protocolRecoveryRetried - }) + // La decisión es de la puerta: un snapshot de este intento más la política de esta + // superficie. El cupo de recuperación de protocolo es del loop, y viaja como ausencia de + // evidencia en el snapshot (ver buildOpenAIAgentGateSnapshot). + const snapshot = buildOpenAIAgentGateSnapshot(attempt, options, protocolRecoveryRetried) + const evaluation = gate(snapshot, openAIAgentGatePolicy()) lastAttempt = attempt lastEvaluation = evaluation upstreamContext = mergePresent(upstreamContext, attempt.metadata) - if (evaluation.accepted) { + if (evaluation.verdict === 'accept') { // 恢复名额已用而本轮仍带拦截/残渣证据 = 第二次事故按原样交付。留一行日志, // 生产环境要能区分"提示被采纳、回合恢复"和"第二次、原样交付"。 if (protocolRecoveryRetried && attempt.toolCalls.length === 0 && @@ -884,23 +891,23 @@ const runOpenAIAgentTurn = async (initialResponse, options = {}) => { ok: true, currentAccount, attempt, - finishReason: evaluation.finishReason, + finishReason: wireFinishReason(attempt, evaluation), attempts: attemptNumber, // 原生接纳但正文被文本 [TOOL CALL] 残渣污染:交付层不转发 visibleText。 suppressVisibleText: evaluation.suppressVisibleText === true } } - // 有丢弃帧时任何拒绝理由都带上名字:invalid_tool_call/required_tool 优先级更高 + // 有丢弃帧时任何拒绝理由都带上名字:tool_error/required_tool 优先级更高 // 时拦截会被盖住,这行日志是生产环境验证拦截确实发生的抓手。 const dropSuffix = (attempt.interceptedToolNames?.length || 0) > 0 ? `; dropped: ${attempt.interceptedToolNames.join(', ')}` : '' - // `detail` distingue en producción los dos invalid_tool_call (toolErrors vs prosa+tools): - // sin él, el incidente 2026-09-06 fue indistinguible por logs. - const detailSuffix = evaluation.detail ? `:${evaluation.detail}` : '' + // La razón ES el detalle desde el ticket 07: el viejo `invalid_tool_call` se partió en + // `tool_error` (errores de herramienta) y `prose_with_tools` (prosa junto a llamadas) — + // los dos distinguibles por logs, como desde el incidente 2026-09-06. logger.warn( - `Agent attempt ${attemptNumber}/${maxAttempts} 被回合门禁拒绝 (${evaluation.retryReason}${detailSuffix}${dropSuffix})`, + `Agent attempt ${attemptNumber}/${maxAttempts} 被回合门禁拒绝 (${evaluation.reason}${dropSuffix})`, 'AGENT' ) if (attempt.streamedVisibleText) { @@ -917,13 +924,15 @@ const runOpenAIAgentTurn = async (initialResponse, options = {}) => { } if (attemptNumber >= maxAttempts || typeof requestSender !== 'function' || options.isClientDisconnected?.()) break - if (evaluation.retryReason === 'intercepted' || evaluation.retryReason === 'malformed_protocol') { + if (PROTOCOL_RECOVERY_REASONS.has(evaluation.reason)) { protocolRecoveryRetried = true } - let retryHint = buildAgentRetryHint(evaluation.retryReason) - // 别的理由(invalid_tool_call/required_tool)盖住拦截时,提示词仍要把关键 - // 事实带上:调用没到客户端。不动优先级、不动名额。 - if (evaluation.retryReason !== 'intercepted' && + let retryHint = openAIRetryHint(evaluation.reason, snapshot, options) + // 别的理由(tool_error/required_tool)盖住拦截时,提示词仍要把关键 + // 事实带上:调用没到客户端。不动优先级、不动名额。No se duplica: el hint local de + // `required_tool` no trae el hecho (el mapa de la puerta sí lo agrega para ese token, y + // por eso ese token no entra por el mapa). + if (evaluation.reason !== REASONS.INTERCEPTED && attempt.toolCalls.length === 0 && (attempt.interceptedToolNames?.length || 0) > 0) { retryHint = `${retryHint}\n${buildAgentRetryHint('intercepted')}` @@ -943,13 +952,16 @@ const runOpenAIAgentTurn = async (initialResponse, options = {}) => { agentRetry: true }) if (!retryResponse?.status || !retryResponse.response) { + // El reenvio de correccion tampoco arranco: su fallo dice por que (cuota 429, transporte + // 503, opaco 502) en vez de aplanarse en un 502 mudo. Sin veredicto se conserva el 502 + // y el code de siempre. return { ok: false, - error: { - status: 502, - message: retryResponse?.message || 'Agent 回合纠正请求失败', - code: 'upstream_retry_failed' - }, + error: openAIErrorShape( + retryResponse?.failure || unclassifiedFailure(502), + retryResponse?.message || 'Agent 回合纠正请求失败', + 'upstream_retry_failed' + ), attempt, attempts: attemptNumber } @@ -985,18 +997,17 @@ const runOpenAIAgentTurn = async (initialResponse, options = {}) => { return { ok: false, currentAccount, - error: exhaustedError(lastAttempt, lastEvaluation?.retryReason), + error: exhaustedError(lastAttempt, lastEvaluation?.reason), attempt: lastAttempt, attempts: attemptsMade } } module.exports = { - NON_RETRYABLE_FINISH_REASONS, normalizeCreatedMetadata, + buildOpenAIAgentGateSnapshot, + openAIAgentGatePolicy, collectOpenAIAgentAttempt, - evaluateOpenAIAgentAttempt, - appendRetryHint, runOpenAIAgentTurn, // chat.js 旧路径共用的原生帧喂入 feedNativeFrame diff --git a/src/utils/request.js b/src/utils/request.js index 3e8cca07..3439138e 100644 --- a/src/utils/request.js +++ b/src/utils/request.js @@ -10,7 +10,8 @@ const { uploadAgentContextFile, buildChatFileDescriptor } = require('./upload.js const { buildRequestHeaders } = require('./header-profile') const { ContextExternalizationError, isTransportInterruption, assertChatChallengeBreakerClosed, - bindChatChallengeContext, chatChallengeFrom, isWafChallengeError, releaseChatProbe + bindChatChallengeContext, chatChallengeFrom, isWafChallengeError, releaseChatProbe, + describeUpstreamFailure, unclassifiedFailure } = require('./upstream-error.js') const { contextPrefixCache, prefixMatches, canonicalHistoryHash } = require('./context-prefix-cache.js') const { @@ -695,7 +696,9 @@ const sendChatRequest = async (body, options = {}) => { return { status: false, response: null, - message: reason + message: reason, + // Fallo local de configuracion, no del cliente: reintentar mas tarde es lo correcto. + failure: unclassifiedFailure(503) } } @@ -743,7 +746,10 @@ const postChatRequest = async (body, options, currentAccount, currentToken, brea return { status: false, response: null, - message: '无法创建或续接 Qwen 会话' + message: '无法创建或续接 Qwen 会话', + // Upstream opaco: los challenges ya se lanzan antes de llegar aqui, asi que esto + // no es "vuelve en un rato" sino "el upstream contesto algo que no entendemos". + failure: unclassifiedFailure(502) } } // 浏览器 referer 为 /c/(在 chat_id 生成后动态设置) @@ -840,10 +846,23 @@ const postChatRequest = async (body, options, currentAccount, currentToken, brea } } + // El veredicto sale con el fallo: los llamadores no tienen que volver a deducir el + // significado de un booleano. Sin respuesta HTTP no hay veredicto que leer — el + // transporte se cayo, y eso es reintentable (503). Con respuesta, decide el clasificador + // (429 del upstream o 502 opaco). + const failure = lastError?.response + ? describeUpstreamFailure(lastError, 502, 503) + : unclassifiedFailure(503) + // 所有尝试失败 — 分类错误 if (lastError && currentAccount?.email) { const hadHttpResponse = !!lastError.response - if (!hadHttpResponse && isRetryableNetworkError(lastError)) { + if (failure.rateLimited) { + // La misma pared que el canal de payload, por el otro canal: la cuenta queda en + // cuota agotada (cooldown por defecto si el upstream no dijo cuanto esperar). + logger.error(`发送聊天请求失败: 账户额度已耗尽`, 'REQUEST', '', lastError.message) + accountManager.recordAccountQuotaExhausted(currentAccount.email, failure.retryAfter) + } else if (!hadHttpResponse && isRetryableNetworkError(lastError)) { // 传输层失败耗尽重试——记 failure,累计可触发 cooldown(PR #112 语义) logger.error( `聊天请求传输失败 (已尝试 ${totalAttempts} 次): ${lastError.message}`, @@ -856,7 +875,10 @@ const postChatRequest = async (body, options, currentAccount, currentToken, brea ) accountManager.recordAccountFailure(currentAccount.email, lastError.code) } else { - // HTTP 4xx/5xx (上游主动拒绝, 账户有效) — 仅刷新 warn 指示, 不影响 cooldown + // HTTP 4xx/5xx (上游主动拒绝, 账户有效) — 仅刷新 warn 指示, 不影响 cooldown。 + // Excepción deliberada y registrada: el 429, que cae en la rama de cuota agotada + // de arriba y sí enfría la cuenta — con el cooldown por defecto cuando el cuerpo + // no trae espera. const status = lastError.response?.status logger.error('发送聊天请求失败', 'REQUEST', '', lastError.message) accountManager.recordAccountError(currentAccount.email, status) @@ -871,7 +893,8 @@ const postChatRequest = async (body, options, currentAccount, currentToken, brea return { status: false, - response: null + response: null, + failure } } diff --git a/src/utils/upstream-error.js b/src/utils/upstream-error.js index 376f9478..86fd1543 100644 --- a/src/utils/upstream-error.js +++ b/src/utils/upstream-error.js @@ -118,6 +118,10 @@ const RATE_LIMIT_MESSAGE_RE = /upper limit for today|reached the upper limit|已 const isRateLimitError = (error) => { if (!error || typeof error !== 'object') return false; if (isWafChallengeError(error)) return false; + // Un 429 del upstream es la cuota dicha por el otro canal: la misma pared, sin cuerpo que + // leer. Clasificarla aqui hace que TODOS los consumidores de este predicado la vean igual + // (el failover dentro de la peticion, la marca de cuenta agotada, el registro de fallos). + if (Number(error.response?.status) === 429) return true; const code = String(error.code || '').toLowerCase(); if (code === RATE_LIMIT_CODE.toLowerCase() || code === QUOTA_LIMIT_CODE) return true; return RATE_LIMIT_MESSAGE_RE.test(String(error.publicMessage || error.message || '')); @@ -192,8 +196,11 @@ const isContextAttachmentError = (error) => String(error?.code || '') === CONTEX */ const rateLimitRetryAfterSeconds = (error) => { const hours = Number(error?.details?.waitHours); - if (!Number.isFinite(hours) || hours <= 0) return null; - return Math.ceil(hours * 3600); + if (Number.isFinite(hours) && hours > 0) return Math.ceil(hours * 3600); + // Canal HTTP: un 429 puede traer la espera como cabecera. Solo segundos — una fecha HTTP + // no se interpreta, porque no hay caso medido que la produzca. + const header = Number(error?.response?.headers?.['retry-after']); + return Number.isFinite(header) && header > 0 ? Math.ceil(header) : null; }; /** @@ -257,6 +264,44 @@ const describeUpstreamFailure = (error, fallbackStatus = 502, overloadedStatus = return { rateLimited: true, overloaded: false, status: 429, retryAfter: rateLimitRetryAfterSeconds(error) }; }; +/** + * El veredicto cuando no hay causa de upstream que leer: nadie clasifico, y el status lo + * elige quien conoce el caso (503 configuracion local, 502 upstream opaco). Existe para que + * la forma del veredicto se defina una sola vez, aqui, y no en cada sitio que la construye. + * @param {number} status - Status que corresponde al caso + * @returns {{rateLimited: boolean, overloaded: boolean, status: number, retryAfter: null}} + */ +const unclassifiedFailure = (status) => ({ + rateLimited: false, + overloaded: false, + status, + retryAfter: null +}); + +/** + * Un veredicto de upstream en la forma de error de cable OpenAI. Vive aqui, junto a los + * constantes de los dos vocabularios, porque la usan tres sitios que no pueden depender + * unos de otros: el controlador de chat (via de excepcion y via de retorno) y el runtime + * de agente, que ya devolvia un error con status y code de este cable. + * @param {{rateLimited: boolean, overloaded: boolean, status: number, retryAfter: number|null}} failure + * @param {string} message - Mensaje para el cliente + * @param {string} [fallbackCode] - `code` cuando el veredicto no trae uno propio + * @returns {{status: number, message: string, code: string, type?: string, retry_after?: number}} + */ +const openAIErrorShape = (failure, message, fallbackCode = 'upstream_error') => { + const shape = { + status: failure.status, + message, + code: failure.rateLimited + ? RATE_LIMIT_OPENAI_TYPE + : (failure.overloaded ? 'upstream_unavailable' : fallbackCode) + }; + if (failure.rateLimited) shape.type = RATE_LIMIT_OPENAI_TYPE; + else if (failure.overloaded) shape.type = 'server_error'; + if (failure.retryAfter !== null) shape.retry_after = failure.retryAfter; + return shape; +}; + /** * Denuncia la cuenta que se quedo sin cuota, para que la rotacion deje de elegirla. * @@ -482,6 +527,8 @@ module.exports = { setChatChallengeClockForTests, rateLimitRetryAfterSeconds, describeUpstreamFailure, + unclassifiedFailure, + openAIErrorShape, noteRateLimitedAccount, ContextExternalizationError, isContextAttachmentError, diff --git a/tests/agent-protocol.test.js b/tests/agent-protocol.test.js index a21b6df6..2728ee89 100644 --- a/tests/agent-protocol.test.js +++ b/tests/agent-protocol.test.js @@ -134,8 +134,8 @@ test('Defect A: thinking deltas pass through even with no role', () => { }) test('finish reasons preserve truncation instead of reporting normal completion', () => { - assert.equal(normalizeOpenAIFinishReason('length', false, true), 'length') - assert.equal(normalizeOpenAIFinishReason(null, false, false), null) + assert.equal(normalizeOpenAIFinishReason('length', true), 'length') + assert.equal(normalizeOpenAIFinishReason(null, false), null) assert.equal(mapAnthropicStopReason('length', false, true), 'max_tokens') assert.equal(mapAnthropicStopReason(null, false, false), null) assert.equal(mapAnthropicStopReason('stop', true, true), 'tool_use') @@ -639,6 +639,45 @@ test('strict non-stream Agent gate returns an HTTP error instead of a fake compl assert.equal(res.statusCode, 502) const payload = JSON.parse(res.output) assert.equal(payload.error.code, 'upstream_agent_turn_incomplete') + // El texto del agotamiento lo sirve el mapa de la puerta (EXHAUSTED_TURN_MESSAGES): no se + // movió con la consolidación del ticket 07. + assert.equal(payload.error.message, '上游连续返回未声明完成状态的文本,已阻止 Agent 将未完成任务误判为结束') + assert.equal(Object.hasOwn(payload, 'choices'), false) +}) + +test('strict non-stream Agent gate: el agotamiento por error de herramienta conserva mensaje y código de cable', async () => { + const toolErrorRound = `data: ${JSON.stringify({ + choices: [{ delta: { phase: 'answer', content: '[TOOL CALL]{"name":"nope","arguments":{}}[END TOOL CALL]' }, finish_reason: 'stop' }] + })}\n\n` + let retries = 0 + const res = createMockResponse() + await handleNonStreamResponse( + res, + Readable.from([toolErrorRound]), + false, + false, + 'qwen-test', + { messages: [{ role: 'user', content: 'finish and verify everything' }] }, + { + has_tools: true, + tool_choice: 'auto', + allowed_tool_names: ['run_tests'], + agent_turn_max_attempts: 2, + sendChatRequest: async () => { + retries += 1 + return { status: true, response: Readable.from([toolErrorRound]) } + } + } + ) + + assert.equal(retries, 1) + assert.equal(res.statusCode, 502) + const payload = JSON.parse(res.output) + // La mitad "errores de herramienta" del viejo invalid_tool_call (hoy `tool_error`): el + // mensaje es el de siempre —ahora servido por el mapa de la puerta— y el código de cable + // no cambió con el renombre del vocabulario. + assert.equal(payload.error.message, '上游连续返回残缺、非法或不存在的工具调用') + assert.equal(payload.error.code, 'invalid_tool_call') assert.equal(Object.hasOwn(payload, 'choices'), false) }) @@ -1225,8 +1264,17 @@ test('interleaved multi-response frames are not merged into a duplicated answer' // --------------------------------------------------------------------------- // 回合门禁放宽开关(默认关闭,严格行为不变) // --------------------------------------------------------------------------- +// Desde el ticket 07 el veredicto de esta superficie lo decide la puerta compartida: estos +// tests afirman contra la puerta VIVA, con el mismo snapshot y la misma política que arma el +// runtime (buildOpenAIAgentGateSnapshot + openAIAgentGatePolicy), no contra copias de sus +// reglas. El viejo `invalid_tool_call` con `detail` se partió en dos tokens: `tool_error` +// (errores de herramienta) y `prose_with_tools` (prosa junto a llamadas). const agentTurnConfig = require('../src/config/index.js') -const { evaluateOpenAIAgentAttempt: evaluateAgentTurn } = require('../src/utils/openai-agent-runtime.js') +const { + buildOpenAIAgentGateSnapshot, + openAIAgentGatePolicy +} = require('../src/utils/openai-agent-runtime.js') +const { gate, REASONS } = require('../src/utils/agent-turn-gate.js') const buildAttempt = (overrides = {}) => ({ upstreamFinishReason: null, @@ -1237,6 +1285,20 @@ const buildAttempt = (overrides = {}) => ({ ...overrides }) +/** + * El veredicto vivo de una ronda: snapshot + política, como en el loop. `protocolRecoverySpent` + * reemplaza al viejo `protocol_recovery_used` — el cupo es del loop y el snapshot es donde se + * expresa (la puerta no tiene cupo). + */ +const evaluateAgentTurn = (attempt, { protocolRecoverySpent = false, ...overrides } = {}) => gate( + buildOpenAIAgentGateSnapshot( + attempt, + { has_tools: true, tool_choice: 'auto', ...overrides }, + protocolRecoverySpent + ), + openAIAgentGatePolicy() +) + const withAgentTurnFlags = (flags, fn) => { const saved = { agentTurnAllowProseWithTools: agentTurnConfig.agentTurnAllowProseWithTools, @@ -1250,7 +1312,7 @@ const withAgentTurnFlags = (flags, fn) => { } } -test('默认严格模式:工具调用附带可见正文仍判为 invalid_tool_call', () => { +test('默认严格模式:工具调用附带可见正文仍判为 prose_with_tools', () => { const attempt = buildAttempt({ toolCalls: [{ id: 'call_1', function: { name: 'read', arguments: '{}' } }], controlKind: 'bare', @@ -1258,12 +1320,12 @@ test('默认严格模式:工具调用附带可见正文仍判为 invalid_tool_ }) withAgentTurnFlags({ agentTurnAllowProseWithTools: false }, () => { assert.deepEqual(evaluateAgentTurn(attempt), { - accepted: false, + verdict: 'retry', finishReason: null, - retryReason: 'invalid_tool_call', - // Ronda NO cortada (sin textChannelCut): el rechazo estricto se mantiene; `detail` - // sólo etiqueta cuál de los dos invalid_tool_call fue, para los logs de producción. - detail: 'prose_with_tools' + // La partición del vocabulario: esta razón es la mitad "prosa + tools" del viejo + // invalid_tool_call, la que el incidente 2026-09-06 no podía distinguir por logs. + reason: 'prose_with_tools', + suppressVisibleText: false }) }) }) @@ -1276,9 +1338,10 @@ test('AGENT_TURN_ALLOW_PROSE_WITH_TOOLS 打开后接受正文与工具调用共 }) withAgentTurnFlags({ agentTurnAllowProseWithTools: true }, () => { assert.deepEqual(evaluateAgentTurn(attempt), { - accepted: true, + verdict: 'accept', finishReason: 'tool_calls', - retryReason: null + reason: null, + suppressVisibleText: false }) }) }) @@ -1289,27 +1352,30 @@ test('打开放宽开关也不会接受非法工具调用', () => { visibleText: '正文' }) withAgentTurnFlags({ agentTurnAllowProseWithTools: true, agentTurnAcceptBareFinal: true }, () => { - assert.equal(evaluateAgentTurn(attempt).accepted, false) - assert.equal(evaluateAgentTurn(attempt).retryReason, 'invalid_tool_call') + assert.equal(evaluateAgentTurn(attempt).verdict, 'retry') + // La otra mitad del viejo invalid_tool_call: el veto por error de herramienta. + assert.equal(evaluateAgentTurn(attempt).reason, REASONS.TOOL_ERROR) }) }) test('默认严格模式:缺少 包装的正文判为 bare', () => { const attempt = buildAttempt({ controlKind: 'bare', visibleText: '已经改完了。' }) withAgentTurnFlags({ agentTurnAcceptBareFinal: false }, () => { - assert.equal(evaluateAgentTurn(attempt).retryReason, 'bare') + assert.equal(evaluateAgentTurn(attempt).reason, REASONS.BARE) }) }) -test('AGENT_TURN_ACCEPT_BARE_FINAL 打开后按 stop 接受,空正文仍判为 bare', () => { +test('AGENT_TURN_ACCEPT_BARE_FINAL 打开后按 stop 接受,空正文仍被拒绝', () => { withAgentTurnFlags({ agentTurnAcceptBareFinal: true }, () => { assert.deepEqual( evaluateAgentTurn(buildAttempt({ controlKind: 'bare', visibleText: '已经改完了。' })), - { accepted: true, finishReason: 'stop', retryReason: null } + { verdict: 'accept', finishReason: 'stop', reason: null, suppressVisibleText: false } ) + // Texto en blanco: parseAgentControlText nunca lo llama `bare` (lo devuelve como + // `empty`), y la regla que sí lo alcanza lo sigue rechazando. assert.equal( - evaluateAgentTurn(buildAttempt({ controlKind: 'bare', visibleText: ' ' })).retryReason, - 'bare' + evaluateAgentTurn(buildAttempt({ controlKind: 'empty', visibleText: ' ' })).reason, + REASONS.EMPTY ) }) }) @@ -1319,8 +1385,8 @@ test('AGENT_TURN_ACCEPT_BARE_FINAL 打开后按 stop 接受,空正文仍判为 // literal en el canal de razonamiento de un turno que el gate acepta, ese texto debe // llegar entero: antes el parser se lo tragaba y la frase salía cortada en el tag. // -// El mismo eco en el canal de respuesta NO llega: el gate estricto lo marca -// invalid_tool_call y, como ya salió texto, cierra con 422 +// El mismo eco en el canal de respuesta NO llega: el gate estricto lo marca tool_error y, +// como ya salió texto, cierra con 422 // upstream_agent_stream_invalidated. Eso es comportamiento deliberado del gate // (tests "默认严格模式" + los switches AGENT_TURN_*), no algo que este cambio toque. // Anotado en deferred-work.md. @@ -1382,7 +1448,7 @@ const AGENT_LEAK = [ ].join('\n') // El mismo payload SIN closer: residuo (forma de leak) pero no protocolo demostrable → // sigue disparando malformed_protocol. Desde el spec narrated-toolcall (2026-09-02) la -// forma CON closer y nombre no declarado es un error duro (unknown_tool → invalid_tool_call). +// forma CON closer y nombre no declarado es un error duro (unknown_tool → tool_error). const AGENT_LEAK_NO_CLOSER = AGENT_LEAK.split('\n')[0] const runAgentTurn = (initialFrames, sendChatRequest, overrides = {}) => runOpenAIAgentTurn( @@ -1405,14 +1471,15 @@ test('OpenAI gate: la narracion envuelta tras drops se rechaza como intercepted, interceptedToolNames: ['Bash'] }) assert.deepEqual(evaluateAgentTurn(attempt), { - accepted: false, + verdict: 'retry', finishReason: null, - retryReason: 'intercepted' + reason: REASONS.INTERCEPTED, + suppressVisibleText: false }) // Nombre agotado el tope compartido: se entrega por las reglas de siempre. - assert.equal(evaluateAgentTurn(attempt, { protocol_recovery_used: true }).accepted, true) + assert.equal(evaluateAgentTurn(attempt, { protocolRecoverySpent: true }).verdict, 'accept') // Sin herramientas en juego, los drops no significan nada. - assert.equal(evaluateAgentTurn(attempt, { has_tools: false }).accepted, true) + assert.equal(evaluateAgentTurn(attempt, { has_tools: false }).verdict, 'accept') }) test('OpenAI gate: drops junto a una llamada aceptada no reintenta (drop especulativo benigno)', () => { @@ -1421,21 +1488,22 @@ test('OpenAI gate: drops junto a una llamada aceptada no reintenta (drop especul interceptedToolNames: ['Bash'] }) assert.deepEqual(evaluateAgentTurn(attempt), { - accepted: true, + verdict: 'accept', finishReason: 'tool_calls', - retryReason: null + reason: null, + suppressVisibleText: false }) }) test('OpenAI gate: el leak de protocolo malformado se rechaza; JSON ordinario no', () => { assert.equal( - evaluateAgentTurn(buildAttempt({ controlKind: 'bare', visibleText: AGENT_LEAK })).retryReason, - 'malformed_protocol' + evaluateAgentTurn(buildAttempt({ controlKind: 'bare', visibleText: AGENT_LEAK })).reason, + REASONS.MALFORMED_PROTOCOL ) // JSON sin clave "arguments" al inicio: respuesta normal, cae en bare. assert.equal( - evaluateAgentTurn(buildAttempt({ controlKind: 'bare', visibleText: '{"name": "results", "count": 3}' })).retryReason, - 'bare' + evaluateAgentTurn(buildAttempt({ controlKind: 'bare', visibleText: '{"name": "results", "count": 3}' })).reason, + REASONS.BARE ) }) @@ -1494,7 +1562,7 @@ test('OpenAI loop: el leak malformado (sin closer) reintenta con su hint y recup assert.match(JSON.stringify(sent[0]), /was NOT executed/) }) -test('OpenAI loop: el leak CON closer y nombre no declarado es error duro (spec narrated-toolcall) → invalid_tool_call, reintenta y recupera', async () => { +test('OpenAI loop: el leak CON closer y nombre no declarado es error duro (spec narrated-toolcall) → tool_error, reintenta y recupera', async () => { const sent = [] const result = await runAgentTurn( [agentAnswerFrame(AGENT_LEAK)], @@ -1508,7 +1576,10 @@ test('OpenAI loop: el leak CON closer y nombre no declarado es error duro (spec assert.equal(result.finishReason, 'tool_calls') assert.equal(sent.length, 1) const hint = JSON.stringify(sent[0]) - assert.match(hint, /invalid, truncated, or unknown tool call/, 'la razon es invalid_tool_call (unknown_tool), no malformed_protocol') + assert.match(hint, /invalid, truncated, or unknown tool call/, 'la razon es tool_error (unknown_tool), no malformed_protocol') + // Y con el constructor compartido el hint nombra el nombre inventado: el único texto que + // hace recuperable un nombre desconocido (cambio declarado del ticket 07). + assert.match(hint, /The tool name\(s\) AskUserQuestion do not exist/) assert.doesNotMatch(hint, /was NOT executed/) }) @@ -1627,8 +1698,6 @@ test('P10: los drops internos no queman el slot que malformed_protocol necesita' // Antes cada frame se hacia push() en index 0 con `+=` → JSON invalido → invalid_tool_call // → retry quemado. Fixtures byte-fieles a scratchpad/capture-foreign.txt (2026-09-01). -const { createNativeToolCallAccumulator: createNativeAccumulatorForIndexPin } = require('../src/utils/tool-prompt.js') - const agentNativeCallFrame = (name, snapshot) => `data: ${JSON.stringify({ choices: [{ delta: { @@ -1931,7 +2000,7 @@ test('OpenAI loop: sin result frames la prosa cierra la llamada nativa; no hay c assert.equal(result.attempt.visibleText.trim(), '') }) -test('OpenAI loop: la ronda de code_interpreter (plataforma) sigue siendo invalid_tool_call con retry, sin corte temprano', async () => { +test('OpenAI loop: la ronda de code_interpreter (plataforma) sigue siendo tool_error con retry, sin corte temprano', async () => { const sent = [] const frames = [ ...CODE_INTERPRETER_SNAPSHOTS.map(snapshot => agentPlatformCallFrame('code_interpreter', snapshot, CODE_INTERPRETER_ID)), @@ -2100,80 +2169,34 @@ test('OpenAI non-stream e2e (F2): texto contaminado → content null; prosa limp assert.equal(clean.message.content, 'Let me check.', 'la prosa limpia previa a la llamada sigue saliendo') }) -// ── chat.js legacy (strict_agent_turn: false): feed nativo, index unico, retry limpio ── - -const legacyToolCallHeaders = (output) => output - .split('\n\n') - .filter(line => line.startsWith('data: ') && line !== 'data: [DONE]') - .map(line => JSON.parse(line.slice(6))) - .flatMap(chunk => (chunk.choices?.[0]?.delta?.tool_calls || [])) -const legacyArgsOf = (deltas, index) => deltas - .filter(call => call.index === index && !call.id) - .map(call => call.function.arguments) - .join('') - -const runLegacyStream = async (frames, options = {}) => { +// Ticket 01 de `.scratch/agent-turn-gate/`: al borrar el camino de herramientas +// inalcanzable quedó a la vista un comportamiento que NO estaba gateado por +// `hasTools`. El reintento por respuesta vacía del loop no-stream reconstruía el +// acumulador nativo y volvía a parsear el texto del reintento — sin condición de +// herramientas —, así que una petición SIN herramientas cuyo primer intento salía +// vacío y cuyo reintento devolvía un bloque `[TOOL CALL]` terminaba entregando +// `tool_calls` (y `finish_reason: "tool_calls"`) a un cliente que nunca declaró +// herramienta alguna. El propio hint del reintento pide ese bloque, así que el +// lazo se cerraba solo. +// +// Este test fija la conducta nueva: una petición sin herramientas nunca recibe +// `tool_calls`. El residuo de protocolo viaja verbatim como contenido, que es lo +// que ya hacían el primer intento de esta misma ruta y el gemelo streaming. +test('OpenAI non-stream sin herramientas: el reintento nunca produce tool_calls fantasma', async () => { + let retries = 0 const res = createMockResponse() - await handleStreamResponse( + await handleNonStreamResponse( res, - Readable.from(frames), + Readable.from(['data: {"choices":[{"delta":{},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n']), false, false, - { messages: [{ role: 'user', content: 'do the task' }] }, - { has_tools: true, strict_agent_turn: false, tool_choice: 'auto', allowed_tool_names: NATIVE_TOOLS, ...options } - ) - return res -} - -test('chat.js legacy stream: una llamada nativa produce exactamente un header tool_calls[0] con los arguments exactos', async () => { - const res = await runLegacyStream([ - ...nativeAgentTurn('Bash', NATIVE_BASH_SNAPSHOTS), - agentNotExistsFrame('Bash'), - AGENT_FINISHED_FRAME, - 'data: {"choices":[{"delta":{},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n' - ]) - assert.doesNotMatch(res.output, /invalid_tool_call/) - const deltas = legacyToolCallHeaders(res.output) - const headers = deltas.filter(call => call.id) - assert.equal(headers.length, 1, 'un snapshot repetido no puede abrir una segunda llamada') - assert.equal(headers[0].index, 0) - assert.equal(headers[0].function.name, 'Bash') - assert.equal(legacyArgsOf(deltas, 0), NATIVE_BASH_ARGS) - assert.match(res.output, /"finish_reason":"tool_calls"/) -}) - -test('chat.js legacy stream: la llamada textual y la nativa no pueden ser ambas tool_calls[0]', async () => { - const res = await runLegacyStream([ - agentAnswerFrame('[TOOL CALL]{"name":"Bash","arguments":{"command":"ls"}}[END TOOL CALL]'), - ...nativeAgentTurn('Bash', NATIVE_BASH_SNAPSHOTS), - agentNotExistsFrame('Bash'), - AGENT_FINISHED_FRAME, - 'data: {"choices":[{"delta":{},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n' - ]) - const deltas = legacyToolCallHeaders(res.output) - const headers = deltas.filter(call => call.id) - assert.deepEqual(headers.map(call => call.function.name), ['Bash', 'Bash']) - assert.deepEqual(headers.map(call => call.index), [0, 1], 'el caller es dueno del unico index monotono') - assert.equal(legacyArgsOf(deltas, 0), '{"command":"ls"}') - assert.equal(legacyArgsOf(deltas, 1), NATIVE_BASH_ARGS) - // El accumulator por si solo sigue numerando desde 0: la unificacion vive en el caller. - const twin = createNativeAccumulatorForIndexPin({ allowedToolNames: NATIVE_TOOLS }) - twin.pushNativeSnapshot({ name: 'Bash', arguments: NATIVE_BASH_ARGS, phase: 'answer' }) - assert.equal(twin.finalize()[0].index, 0) -}) - -test('chat.js legacy stream: el retry de compensacion recrea parser y accumulator — el fragmento de la ronda 1 no reaparece', async () => { - let sent = 0 - const res = await runLegacyStream( - [ - agentAnswerFrame('[TOOL CALL]{"name":"Bash","arg'), - 'data: {"choices":[{"delta":{},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n' - ], + 'qwen-test', + { messages: [{ role: 'user', content: 'hola' }] }, { - tool_choice: 'required', + has_tools: false, sendChatRequest: async () => { - sent += 1 + retries += 1 return { status: true, response: Readable.from([ @@ -2184,12 +2207,10 @@ test('chat.js legacy stream: el retry de compensacion recrea parser y accumulato } } ) - assert.equal(sent, 1, 'tool_choice=required sin llamada dispara la compensacion') - assert.doesNotMatch(res.output, /invalid_tool_call/, 'el fragmento de la ronda 1 contamino el parser de la ronda 2') - const deltas = legacyToolCallHeaders(res.output) - const headers = deltas.filter(call => call.id) - assert.equal(headers.length, 1) - assert.equal(headers[0].function.name, 'Bash') - assert.equal(legacyArgsOf(deltas, headers[0].index), '{"command":"ls"}') - assert.match(res.output, /"finish_reason":"tool_calls"/) + + assert.equal(retries, 1, 'la respuesta vacia dispara el reintento de compensacion') + const body = JSON.parse(res.output) + assert.equal(body.choices[0].message.tool_calls, undefined, + 'un cliente que no declaro herramientas no puede recibir tool_calls') + assert.notEqual(body.choices[0].finish_reason, 'tool_calls') }) diff --git a/tests/agent-resend-verdict.test.js b/tests/agent-resend-verdict.test.js new file mode 100644 index 00000000..d7c520b5 --- /dev/null +++ b/tests/agent-resend-verdict.test.js @@ -0,0 +1,76 @@ +// El reenvio de correccion del runtime de agente (una ronda rechazada -> segundo envio con el +// hint) devolvia un 502 opaco cuando ese segundo envio no arrancaba: cuota o transporte caido +// quedaban indistinguibles de un servidor roto. El fallo del reenvio trae veredicto desde el +// modulo de request, y el runtime lo propaga tal cual para que el controlador lo escriba. +const test = require('node:test') +const assert = require('node:assert/strict') +const { Readable } = require('node:stream') + +process.env.API_KEY = 'agent-resend-verdict-test-key' +process.env.DATA_SAVE_MODE = 'none' +process.env.ACCOUNTS = '' +process.env.ENABLE_CLI = 'false' +process.env.ENABLE_FILE_LOG = 'false' +process.env.PROXY_URL = '' + +const modelsMap = require('../src/models/models-map.js') +modelsMap.getLatestModels = async () => { throw new Error('offline test: no model fetch') } +const { runOpenAIAgentTurn } = require('../src/utils/openai-agent-runtime.js') + +test.after(() => { require('../src/utils/account.js').destroy() }) + +const frame = (payload) => `data: ${JSON.stringify(payload)}\n\n` +/** Una ronda sin tool call y sin cierre: la puerta la rechaza y pide un reenvio. */ +const unfinishedTurn = () => [frame({ choices: [{ delta: { phase: 'answer', content: 'a medias' }, finish_reason: null }] })] + +const options = (sendChatRequest) => ({ + has_tools: true, + tool_choice: 'auto', + allowed_tool_names: ['read_file'], + agent_turn_max_attempts: 3, + upstream_request_body: { messages: [{ role: 'user', content: 'do the task' }] }, + currentAccount: { email: 'resend@example.invalid', token: 'resend-test-token' }, + sendChatRequest +}) + +const run = (sendChatRequest) => runOpenAIAgentTurn(Readable.from(unfinishedTurn()), options(sendChatRequest)) + +test('un reenvio que no arranca por cuota sale 429, no un 502 opaco', async () => { + let calls = 0 + const result = await run(async () => { + calls += 1 + return { + status: false, + response: null, + failure: { rateLimited: true, overloaded: false, status: 429, retryAfter: 3600 } + } + }) + + assert.equal(calls, 1, 'el reenvio de correccion se intento una vez') + assert.equal(result.ok, false) + assert.equal(result.error.status, 429) + assert.equal(result.error.type, 'insufficient_quota') + assert.equal(result.error.code, 'insufficient_quota') + assert.equal(result.error.retry_after, 3600) +}) + +test('un reenvio que no arranca por transporte sale 503', async () => { + const result = await run(async () => ({ + status: false, + response: null, + failure: { rateLimited: false, overloaded: false, status: 503, retryAfter: null } + })) + + assert.equal(result.ok, false) + assert.equal(result.error.status, 503) + assert.equal(result.error.code, 'upstream_retry_failed', 'sin veredicto propio conserva el code de siempre') +}) + +test('un fallo sin veredicto conserva el 502 y el code de siempre', async () => { + const result = await run(async () => ({ status: false, response: null, message: 'algo se rompio' })) + + assert.equal(result.ok, false) + assert.equal(result.error.status, 502) + assert.equal(result.error.code, 'upstream_retry_failed') + assert.equal(result.error.message, 'algo se rompio') +}) diff --git a/tests/agent-turn-budget.test.js b/tests/agent-turn-budget.test.js new file mode 100644 index 00000000..b340eae9 --- /dev/null +++ b/tests/agent-turn-budget.test.js @@ -0,0 +1,121 @@ +'use strict' + +/** + * Ticket 04 de `.scratch/agent-turn-gate/`: un solo significado de "max attempts". + * + * El runtime OpenAI aplicaba un piso de 2 sobre el valor que le pasaran — pedir 1 daba 2 + * intentos en silencio — mientras las superficies Anthropic aplicaban piso 1. La misma + * configuración significaba dos cosas. Ahora "max attempts" es el total de generaciones del + * upstream por petición de cliente, contando la primera, y pedir 1 da un intento. + * + * Lo observable es cuántas veces se le pidió una generación al upstream: el sender es un + * guion que cuenta. + */ + +// Config se congela al importarse y el presupuesto sale de ahí cuando la petición no trae +// uno, así que el pin va antes de cualquier require que lo arrastre. +process.env.AGENT_TURN_MAX_ATTEMPTS = '2' +process.env.DATA_SAVE_MODE = 'none' +process.env.ACCOUNTS = '' +process.env.LEGACY_REASONING_IN_CONTENT = 'false' +process.env.LOG_LEVEL = 'error' +process.env.ENABLE_FILE_LOG = 'false' + +const test = require('node:test') +const assert = require('node:assert/strict') +const { Readable } = require('node:stream') +const { handleStreamResponse } = require('../src/controllers/chat.js') +const { handleAnthropicNonStream } = require('../src/controllers/anthropic.js') + +test.after(() => { + try { require('../src/utils/account.js').destroy() } catch (_) { /* nada que limpiar */ } +}) + +const frame = (payload) => `data: ${JSON.stringify(payload)}\n\n` +const stop = () => frame({ choices: [{ delta: {}, finish_reason: 'stop' }] }) + 'data: [DONE]\n\n' + +/** Ronda que no cumple `required`: prosa sin llamada. Siempre rechazada, nunca recupera. */ +const proseRound = () => Readable.from([ + frame({ choices: [{ delta: { phase: 'answer', content: 'Voy a revisar el repositorio.' } }] }), + stop() +]) + +const createStreamResponse = () => ({ + statusCode: 200, + output: '', + headers: {}, + headersSent: false, + writableEnded: false, + set(headers) { Object.assign(this.headers, headers); return this }, + status(code) { this.statusCode = code; return this }, + write(chunk) { this.headersSent = true; this.output += String(chunk); return true }, + end(chunk = '') { this.output += String(chunk); this.writableEnded = true }, + json(payload) { this.body = payload; this.writableEnded = true; return this } +}) + +/** Sender guion: cuenta generaciones y siempre devuelve otra ronda de prosa. */ +const countingSender = () => { + const fn = async () => { fn.sends += 1; return { status: true, response: proseRound() } } + fn.sends = 0 + return fn +} + +const runOpenAI = async (budget) => { + const res = createStreamResponse() + const sender = countingSender() + const body = { messages: [{ role: 'user', content: 'do the task' }] } + await handleStreamResponse(res, proseRound(), false, false, body, { + has_tools: true, + tool_choice: 'required', + allowed_tool_names: ['Bash'], + sendChatRequest: sender, + upstream_request_body: body, + currentAccount: null, + upstreamOptions: {}, + agent_turn_max_attempts: budget + }) + return sender.sends + 1 // la primera generación no pasa por el sender +} + +const runAnthropic = async (overrides = {}) => { + const sender = countingSender() + await handleAnthropicNonStream( + createStreamResponse(), + { + message_id: 'msg_budget', + model: 'qwen-test', + hasTools: true, + toolChoice: 'required', + allowedToolNames: ['Bash'], + requestBody: { messages: [{ role: 'user', content: 'do the task' }] }, + sendRequest: sender, + historyToolCalls: [], + upstreamOptions: {}, + ...overrides + }, + proseRound() + ) + return sender.sends + 1 +} + +test('presupuesto 1 significa un intento: la ronda rechazada no se reintenta', async () => { + assert.equal(await runOpenAI(1), 1, + 'pedir 1 daba 2 generaciones del upstream: el piso de 2 del runtime OpenAI') +}) + +test('presupuesto 2 significa dos intentos: un reintento y se acabó', async () => { + assert.equal(await runOpenAI(2), 2) +}) + +test('la superficie Anthropic honra el mismo presupuesto de configuración', async () => { + assert.equal(await runAnthropic(), 2, + 'config en 2 = dos generaciones, la misma cuenta que en la superficie OpenAI') +}) + +test('la superficie Anthropic no tiene override por petición', async () => { + // No existe ese canal: el ctx no lo lee y el presupuesto sale de config. Se pasa uno + // igual para fijar que NO se empieza a honrar algo que nadie definió — el día que exista + // un override de Anthropic, este test es el que obliga a decidirlo en vez de heredarlo. + assert.equal(await runAnthropic({ agent_turn_max_attempts: 1 }), 2, + 'un override colado en el ctx se ignora: sigue mandando config') +}) diff --git a/tests/agent-turn-corpus.scenarios.js b/tests/agent-turn-corpus.scenarios.js new file mode 100644 index 00000000..355d9639 --- /dev/null +++ b/tests/agent-turn-corpus.scenarios.js @@ -0,0 +1,729 @@ +'use strict' + +/** + * Corpus de caracterización del gate de turno agéntico (issue 03, .scratch/agent-turn-gate). + * + * Congela QUÉ HACE HOY el código de aceptación de turno en las cuatro células de la + * matriz — superficie × modo de streaming — para que el refactor de los tickets 04..07 + * (una sola puerta) tenga que aparecer como un diff nombrado contra este archivo. El + * baseline versionado lo escribe tools/dev-probes/record-agent-turn-corpus.js; el test + * permanente (tests/agent-turn-corpus.test.js) reproduce el corpus y afirma cada + * entrada grabada. + * + * Este módulo NO lleva sufijo `.test.js` a propósito: no es un archivo de test (el gate + * de conteo lista tests/*.test.js), es el escenario compartido entre el grabador y el + * test. No introduce ninguna costura nueva: conduce los handlers ya exportados + * (`handleStreamResponse` / `handleNonStreamResponse` con `has_tools: true`, y + * `handleAnthropicStream` / `handleAnthropicNonStream`) inyectando un sender con guion + * por las opciones que la producción ya usa (`options.sendChatRequest` / `ctx.sendRequest`), + * con una respuesta falsa. Sin red, sin cuentas, sin login: el sender es un guion. + * + * Lo grabado por escenario es lo observable, nunca un interno: status terminal, frames o + * cuerpo entregados, si el reintento se disparó y cuántos envíos al upstream hizo, y el + * texto del hint que viajó al modelo (el contrato de cara al modelo es lo primero que + * deriva en un refactor). + * + * Los tokens de razón NO se graban: son internos y sólo se manifiestan a través de su + * hint. Un renombre de token que deje el hint igual es invisible acá — y también lo es + * para el cliente, que es lo que este corpus protege. + * + * `applicable` y `targets` son la DECLARACIÓN de intención del escenario, no evidencia: + * viajan a la fila para documentarla, pero las copia la declaración, no una medición. La + * evidencia son las columnas observadas (status, envíos, frames servidos, hints, + * entregado), y los invariantes que las cruzan viven en `corpusViolations`. + */ + +// Pines de política ANTES de cualquier require que arrastre config/index.js (que hace +// dotenv.config() y congela el entorno al cargarse). El baseline no puede depender de los +// defaults de la máquina que lo corre: sin esto, un LEGACY_REASONING_IN_CONTENT=true en el +// shell del operador reescribiría el baseline. +process.env.AGENT_TURN_MAX_ATTEMPTS = '3' +process.env.AGENT_TURN_ALLOW_PROSE_WITH_TOOLS = 'false' +process.env.AGENT_TURN_ACCEPT_BARE_FINAL = 'false' +process.env.LEGACY_REASONING_IN_CONTENT = 'false' +// Sin cuentas que cargar: en DATA_SAVE_MODE=none un ACCOUNTS no vacío dispara logins reales +// al importar utils/account.js (que los controllers cargan). +process.env.DATA_SAVE_MODE = 'none' +process.env.ACCOUNTS = '' +// Ruido, no comportamiento: el corpus imprime su propia tabla y no debe escribir logs a disco. +process.env.LOG_LEVEL = 'error' +process.env.ENABLE_FILE_LOG = 'false' + +const { handleStreamResponse, handleNonStreamResponse } = require('../src/controllers/chat.js') +const { handleAnthropicStream, handleAnthropicNonStream } = require('../src/controllers/anthropic.js') + +// ───────────────────────────── herramientas ───────────────────────────── + +/** Las mismas tres herramientas en las cuatro células: el corpus compara superficies. */ +const ALLOWED_TOOL_NAMES = ['Read', 'Bash', 'Edit'] +const TOOL_SCHEMAS = { + Read: { type: 'object', properties: { file_path: { type: 'string' } }, required: ['file_path'] }, + Bash: { type: 'object', properties: { command: { type: 'string' } }, required: ['command'] }, + Edit: { type: 'object', properties: { file_path: { type: 'string' } }, required: ['file_path'] } +} + +const BASE_PROMPT = 'CORPUS-BASE-PROMPT: do the task' + +/** Marcador canónico del protocolo de texto: el mismo que enseña el prompt de agente. */ +const textCall = (name, argsJson) => `[TOOL CALL]{"name":"${name}","arguments":${argsJson}}[END TOOL CALL]` +const READ_CALL = textCall('Read', '{"file_path":"a.txt"}') + +// ─────────────────────────── builders de frames ─────────────────────────── + +const sse = (payload) => `data: ${JSON.stringify(payload)}\n\n` + +const answer = (content) => sse({ + choices: [{ delta: { phase: 'answer', content }, finish_reason: null }] +}) +const think = (content) => sse({ + choices: [{ delta: { phase: 'think', content }, finish_reason: null }] +}) +/** + * Frame `role:function` con el que la plataforma devuelve el resultado de una llamada (o su + * ausencia). Es la forma que necesita el escenario `intercepted`: sin este frame no hay + * nada que el normalizador pueda descartar. + */ +const droppedResult = (name) => sse({ + choices: [{ + delta: { + role: 'function', phase: 'answer', status: 'typing', name, + content: `Tool ${name} does not exists.` + }, + finish_reason: null + }] +}) + +/** Terminador de ronda: finish_reason=stop + [DONE]. */ +const STOP = 'data: {"choices":[{"delta":{},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n' + +/** Ronda de recuperación compartida: una llamada limpia que satisface cualquier required. */ +const RECOVERY_ROUND = [answer(READ_CALL), STOP] + +/** + * Upstream de una ronda. Generador crudo, no Readable.from: un corte del canal de texto + * destruye el upstream y el generador se detiene donde tocó (Readable.from precargaría). + * `served` cuenta los frames que el consumidor llegó a tirar: es el observable que hace + * falsable "la guarda abortó el intento a mitad de stream" (si nadie corta, served = total). + */ +const framesStream = (frames) => { + const stream = (async function* () { + for (const frame of frames) { + stream.served += 1 + yield frame + } + })() + stream.served = 0 + stream.total = frames.length + return stream +} + +// ────────────────────────────── guion del sender ────────────────────────────── + +/** + * Sender con guion: la primera ronda llega como `upstream` al handler; cada llamada + * posterior consume la siguiente ronda del guion. Agotado el guion devuelve + * `{status:false}` (el camino "el reintento no arrancó"). + */ +const scriptedSender = (retryRounds) => { + const queue = [...retryRounds] + const fn = async (body) => { + fn.calls.push(body) + const round = queue.shift() + return round ? { status: true, response: framesStream(round) } : { status: false } + } + fn.calls = [] + return fn +} + +/** El hint que viajó al upstream: lo que el reintento AÑADIÓ al último mensaje del cuerpo. */ +const hintOf = (sentBody, baseBody) => { + const last = Array.isArray(sentBody?.messages) ? sentBody.messages[sentBody.messages.length - 1] : null + const baseLast = Array.isArray(baseBody?.messages) ? baseBody.messages[baseBody.messages.length - 1] : null + const sent = typeof last?.content === 'string' ? last.content : JSON.stringify(last?.content ?? null) + const base = typeof baseLast?.content === 'string' ? baseLast.content : '' + const added = base && sent.startsWith(base) ? sent.slice(base.length) : sent + // Se pela sólo el separador que ambas superficies anteponen. El encabezado + // '# Tool-call retry' de la superficie Anthropic NO se pela: es texto que viaja al modelo, + // y pelarlo dejaría fuera del baseline cualquier cambio en ese contrato. + return added.replace(/^\n\n/, '') +} + +// ─────────────────────────── respuestas falsas ─────────────────────────── + +const createStreamResponse = () => ({ + output: '', + headers: {}, + headersSent: false, + writableEnded: false, + destroyed: false, + statusCode: 200, + set(headers) { Object.assign(this.headers, headers); return this }, + status(code) { this.statusCode = code; return this }, + write(chunk) { this.headersSent = true; this.output += String(chunk); return true }, + end(chunk = '') { if (chunk) this.write(chunk); this.writableEnded = true }, + json(value) { + this.headersSent = true + this.output += JSON.stringify(value) + this.writableEnded = true + return this + } +}) + +// ──────────────────── extracción del resultado observable ──────────────────── + +const parseSseFrames = (output) => output + .split('\n\n') + .filter(Boolean) + .map((chunk) => { + const lines = chunk.split('\n') + const eventLine = lines.find(line => line.startsWith('event: ')) + const dataLine = lines.find(line => line.startsWith('data: ')) + return { + event: eventLine ? eventLine.slice(7).trim() : null, + data: dataLine ? dataLine.slice(6) : null + } + }) + .filter(frame => frame.data !== null) + .map((frame) => { + // [DONE] no es JSON: se marca antes de parsear para que no caiga en `unparsed`. + if (frame.data === '[DONE]') return { event: frame.event, payload: '[DONE]' } + try { + return { event: frame.event, payload: JSON.parse(frame.data) } + } catch (_) { + return { event: frame.event, unparsed: frame.data } + } + }) + +// Los ids se acuñan por respuesta (uuid/hex): se graba su FORMA, no su valor, para que el +// baseline no derive en cada corrida. La forma es contrato de cable (prefijos call_/toolu_). +const ID_FORMS = [ + [/^chatcmpl-[0-9a-f-]{36}$/, 'chatcmpl-'], + [/^msg_[0-9a-f]{24}$/, 'msg_'], + [/^call_[0-9a-f]{24}$/, 'call_'], + [/^toolu_[0-9a-f]{24}$/, 'toolu_'] +] +const scrubId = (id) => { + if (typeof id !== 'string') return id ?? null + const match = ID_FORMS.find(([pattern]) => pattern.test(id)) + return match ? match[1] : id +} + +/** Frames SSE del cable OpenAI, normalizados a una lista ordenada y estable. */ +const openaiStreamOutcome = (res) => { + const items = [] + for (const frame of parseSseFrames(res.output)) { + if (frame.unparsed !== undefined) { items.push({ kind: 'unparsed', raw: frame.unparsed }); continue } + const payload = frame.payload + if (payload === '[DONE]') { items.push({ kind: 'done' }); continue } + if (payload.error) { + items.push({ + kind: 'error', + code: payload.error.code ?? null, + type: payload.error.type ?? null, + message: payload.error.message ?? null + }) + continue + } + if (payload.usage) items.push({ kind: 'usage' }) + const choice = payload.choices?.[0] + if (!choice) continue + const delta = choice.delta || {} + if (delta.role) items.push({ kind: 'role', role: delta.role }) + if (delta.reasoning_content) items.push({ kind: 'reasoning', text: delta.reasoning_content }) + if (delta.content) items.push({ kind: 'content', text: delta.content }) + for (const piece of delta.tool_calls || []) { + if (piece.function?.name) { + items.push({ kind: 'tool_call', index: piece.index ?? null, id: scrubId(piece.id), name: piece.function.name }) + } else if (typeof piece.function?.arguments === 'string' && piece.function.arguments) { + items.push({ kind: 'tool_call_args', index: piece.index ?? null, arguments: piece.function.arguments }) + } + } + if (choice.finish_reason) items.push({ kind: 'finish', reason: choice.finish_reason }) + } + return items +} + +/** Cuerpo JSON del cable OpenAI, normalizado. */ +const openaiJsonOutcome = (res) => { + const body = JSON.parse(res.output) + if (body.error) { + return [{ + kind: 'error', + code: body.error.code ?? null, + type: body.error.type ?? null, + message: body.error.message ?? null + }] + } + const items = [] + const choice = body.choices?.[0] + const message = choice?.message || {} + if (message.reasoning_content) items.push({ kind: 'reasoning', text: message.reasoning_content }) + if (message.content) items.push({ kind: 'content', text: message.content }) + for (const call of message.tool_calls || []) { + items.push({ + kind: 'tool_call', + index: call.index ?? null, + id: scrubId(call.id), + name: call.function?.name ?? null, + arguments: call.function?.arguments ?? null + }) + } + items.push({ kind: 'finish', reason: choice?.finish_reason ?? null }) + if (body.usage) items.push({ kind: 'usage' }) + return items +} + +/** Eventos SSE del cable Anthropic, normalizados (sin ping: es ruido de cadencia, no contenido). */ +const anthropicStreamOutcome = (res) => { + const items = [] + for (const frame of parseSseFrames(res.output)) { + if (frame.unparsed !== undefined) { items.push({ kind: 'unparsed', raw: frame.unparsed }); continue } + const payload = frame.payload + switch (payload.type) { + case 'message_start': + items.push({ kind: 'message_start' }) + break + case 'content_block_start': { + const block = payload.content_block || {} + if (block.type === 'tool_use') { + items.push({ kind: 'tool_use_start', index: payload.index, id: scrubId(block.id), name: block.name }) + } else { + items.push({ kind: `${block.type}_block_start`, index: payload.index }) + } + break + } + case 'content_block_delta': { + const delta = payload.delta || {} + if (delta.type === 'text_delta') items.push({ kind: 'content', text: delta.text }) + else if (delta.type === 'thinking_delta') items.push({ kind: 'thinking', text: delta.thinking }) + else if (delta.type === 'input_json_delta') { + items.push({ kind: 'tool_use_args', index: payload.index, partial_json: delta.partial_json }) + } else if (delta.type === 'signature_delta') items.push({ kind: 'signature', index: payload.index }) + else items.push({ kind: 'unknown_delta', delta_type: delta.type ?? null }) + break + } + case 'content_block_stop': + items.push({ kind: 'block_stop', index: payload.index }) + break + case 'message_delta': + items.push({ kind: 'stop_reason', stop_reason: payload.delta?.stop_reason ?? null }) + break + case 'message_stop': + items.push({ kind: 'message_stop' }) + break + case 'error': + items.push({ + kind: 'error', + error_type: payload.error?.type ?? null, + message: payload.error?.message ?? null + }) + break + case 'ping': + break + default: + items.push({ kind: payload.type }) + } + } + return items +} + +/** Cuerpo JSON del cable Anthropic, normalizado. */ +const anthropicJsonOutcome = (res) => { + const body = JSON.parse(res.output) + if (body.type === 'error') { + return [{ + kind: 'error', + error_type: body.error?.type ?? null, + message: body.error?.message ?? null + }] + } + const items = [] + for (const block of body.content || []) { + if (block.type === 'text') items.push({ kind: 'content', text: block.text }) + else if (block.type === 'thinking') items.push({ kind: 'thinking', text: block.thinking }) + else if (block.type === 'tool_use') { + items.push({ kind: 'tool_use', id: scrubId(block.id), name: block.name, input: block.input }) + } else items.push({ kind: 'unknown_block', block_type: block.type ?? null }) + } + items.push({ kind: 'stop_reason', stop_reason: body.stop_reason ?? null }) + return items +} + +// ───────────────────────────── las cuatro células ───────────────────────────── + +const baseRequestBodies = () => ({ + openai: { messages: [{ role: 'user', content: BASE_PROMPT }] }, + anthropic: { messages: [{ role: 'user', content: BASE_PROMPT }] } +}) + +/** Las opciones que la producción arma para el controlador OpenAI, con el sender inyectado. */ +const openAiOptions = (scenario, sender, body) => ({ + has_tools: true, + tool_choice: scenario.toolChoice || 'auto', + allowed_tool_names: ALLOWED_TOOL_NAMES, + tool_schemas: TOOL_SCHEMAS, + sendChatRequest: sender, + upstream_request_body: body, + currentAccount: null, + upstreamOptions: {} +}) + +/** El ctx que la producción arma para el controlador Anthropic, con el sender inyectado. */ +const anthropicCtx = (scenario, sender, body) => ({ + message_id: 'msg_corpus', + model: 'qwen-corpus', + hasTools: true, + toolChoice: scenario.toolChoice || 'auto', + requestBody: body, + allowedToolNames: ALLOWED_TOOL_NAMES, + toolSchemas: TOOL_SCHEMAS, + sendRequest: sender, + historyToolCalls: [], + upstreamOptions: {} +}) + +const SURFACES = [ + { + id: 'openai.stream', + family: 'openai', + label: 'OpenAI /v1/chat/completions, stream', + run: async (scenario, sender) => { + const res = createStreamResponse() + const body = baseRequestBodies().openai + const upstream = framesStream(scenario.rounds[0]) + await handleStreamResponse(res, upstream, true, false, body, openAiOptions(scenario, sender, body)) + return { status: res.statusCode, items: openaiStreamOutcome(res), baseBody: body, upstream } + } + }, + { + id: 'openai.nonstream', + family: 'openai', + label: 'OpenAI /v1/chat/completions, non-stream', + run: async (scenario, sender) => { + const res = createStreamResponse() + const body = baseRequestBodies().openai + const upstream = framesStream(scenario.rounds[0]) + await handleNonStreamResponse(res, upstream, true, false, 'qwen-corpus', body, openAiOptions(scenario, sender, body)) + return { status: res.statusCode, items: openaiJsonOutcome(res), baseBody: body, upstream } + } + }, + { + id: 'anthropic.stream', + family: 'anthropic', + label: 'Anthropic /v1/messages, stream', + run: async (scenario, sender) => { + const res = createStreamResponse() + const ctx = anthropicCtx(scenario, sender, baseRequestBodies().anthropic) + const upstream = framesStream(scenario.rounds[0]) + await handleAnthropicStream(res, ctx, upstream) + return { status: res.statusCode, items: anthropicStreamOutcome(res), baseBody: ctx.requestBody, upstream } + } + }, + { + id: 'anthropic.nonstream', + family: 'anthropic', + label: 'Anthropic /v1/messages, non-stream', + run: async (scenario, sender) => { + const res = createStreamResponse() + const ctx = anthropicCtx(scenario, sender, baseRequestBodies().anthropic) + const upstream = framesStream(scenario.rounds[0]) + await handleAnthropicNonStream(res, ctx, upstream) + return { status: res.statusCode, items: anthropicJsonOutcome(res), baseBody: ctx.requestBody, upstream } + } + } +] + +// ─────────────────────────────── el corpus ─────────────────────────────── +// +// `targets` es la lista de tokens de razón que el escenario pretende cubrir EN ESA +// superficie, en orden de ronda (vacía = el escenario es un "aceptar", no un reintento). +// `applicable: false` marca que el token no existe en esa superficie; los frames se +// conducen igual y se graba lo que la superficie HACE con ellos, porque "aquí ese token +// no existe, y esto es lo que ocurre en su lugar" también es parte de lo congelado — +// omitir la fila dejaría creer que la superficie se comporta como la otra. + +const NARRATION = 'The Bash tool seems unavailable in this environment, so the task cannot continue.' +const UNEXECUTED_ACTION = 'I will read the file now and then summarise what it contains.' +const PLAIN_PROSE = 'The answer is 42, and no tool is needed for it.' +const BROKEN_SIBLING = textCall('WebFetch', '{"url":"https://example.com"}') +const MALFORMED_PROTOCOL_LEAK = '{"name": "RunCommand", "arguments": {"command": "ls -la"}}' + +const SCENARIOS = [ + { + id: 'accept_tool_call', + group: 'accept', + title: 'a clean tool call', + targets: { 'openai.stream': [], 'openai.nonstream': [], 'anthropic.stream': [], 'anthropic.nonstream': [] }, + applicable: { openai: true, anthropic: true }, + rounds: [[answer(READ_CALL), STOP], RECOVERY_ROUND, RECOVERY_ROUND] + }, + { + id: 'accept_final_answer', + group: 'accept', + title: 'final answer with no tool call', + targets: { 'openai.stream': [], 'openai.nonstream': [], 'anthropic.stream': [], 'anthropic.nonstream': [] }, + applicable: { openai: true, anthropic: true }, + rounds: [[answer('All requested work is complete.'), STOP], RECOVERY_ROUND, RECOVERY_ROUND] + }, + { + id: 'required_tool', + group: 'retry', + title: 'tool_choice requires a call, the round has none', + targets: { 'openai.stream': ['required_tool'], 'openai.nonstream': ['required_tool'], 'anthropic.stream': ['required_tool'], 'anthropic.nonstream': ['required_tool'] }, + applicable: { openai: true, anthropic: true }, + toolChoice: 'required', + // Ronda sin texto visible a propósito: con una respuesta final envuelta, el texto ya + // salió al cliente y la guarda de stream invalidado (422) veta el reintento en la + // superficie OpenAI stream ANTES de que el gate decida — la celda dejaría de cubrir + // `required_tool`. Sólo piensa: el gate mira `required` antes que `empty`, así que la + // razón sigue siendo la que el escenario persigue (y el hint lo confirma). + rounds: [[think('The user wants a tool run, but I should answer in prose instead.'), STOP], RECOVERY_ROUND, RECOVERY_ROUND] + }, + { + id: 'tool_error', + group: 'retry', + title: 'a malformed tool call', + targets: { 'openai.stream': ['tool_error'], 'openai.nonstream': ['tool_error'], 'anthropic.stream': ['tool_error'], 'anthropic.nonstream': ['tool_error'] }, + applicable: { openai: true, anthropic: true }, + // Tercera herramienta declarada, nombre inexistente: error DURO del parser (unknown_tool), + // el mismo que distingue "el modelo inventó un nombre" de "el protocolo se rompió". + rounds: [[answer(textCall('WebFetch', '{"url":"https://example.com"}')), STOP], RECOVERY_ROUND, RECOVERY_ROUND] + }, + { + id: 'prose_with_tools', + group: 'retry', + title: 'prose alongside a parsed tool call', + // Celda de asimetría deliberada (policy `proseWithTools`): OpenAI reintenta, Anthropic + // acepta. En Anthropic la fila SÍ aplica — su expectativa es "aceptar y entregar". + targets: { 'openai.stream': ['prose_with_tools'], 'openai.nonstream': ['prose_with_tools'], 'anthropic.stream': [], 'anthropic.nonstream': [] }, + applicable: { openai: true, anthropic: true }, + rounds: [[answer(`Sure, let me look at that file.\n\n${READ_CALL}`), STOP], RECOVERY_ROUND, RECOVERY_ROUND] + }, + { + id: 'intercepted', + group: 'retry', + title: 'a dropped role:function frame with no call', + targets: { 'openai.stream': ['intercepted'], 'openai.nonstream': ['intercepted'], 'anthropic.stream': ['intercepted'], 'anthropic.nonstream': ['intercepted'] }, + applicable: { openai: true, anthropic: true }, + rounds: [[droppedResult('Read'), answer(NARRATION), STOP], RECOVERY_ROUND, RECOVERY_ROUND] + }, + { + id: 'malformed_protocol', + group: 'retry', + title: 'orphan protocol residue in the visible text', + targets: { 'openai.stream': ['malformed_protocol'], 'openai.nonstream': ['malformed_protocol'], 'anthropic.stream': ['malformed_protocol'], 'anthropic.nonstream': ['malformed_protocol'] }, + applicable: { openai: true, anthropic: true }, + // Payload pelado al inicio y SIN cierre: el gate de rescate lo rechaza en blando (no es + // error del parser), así que queda como residuo huérfano en la prosa visible. + rounds: [[answer(MALFORMED_PROTOCOL_LEAK), STOP], RECOVERY_ROUND, RECOVERY_ROUND] + }, + { + id: 'thought_tool_call', + group: 'retry', + title: 'a call leaked into the reasoning phase', + targets: { 'openai.stream': [], 'openai.nonstream': [], 'anthropic.stream': ['thought_tool_call'], 'anthropic.nonstream': ['thought_tool_call'] }, + applicable: { openai: false, anthropic: true }, + rounds: [[think(READ_CALL), answer('I have read the file and here is my summary.'), STOP], RECOVERY_ROUND, RECOVERY_ROUND] + }, + { + id: 'missing_tool', + group: 'retry', + title: 'prose that describes an action without calling a tool', + targets: { 'openai.stream': [], 'openai.nonstream': [], 'anthropic.stream': ['missing_tool'], 'anthropic.nonstream': ['missing_tool'] }, + applicable: { openai: false, anthropic: true }, + rounds: [[answer(UNEXECUTED_ACTION), STOP], RECOVERY_ROUND, RECOVERY_ROUND] + }, + { + id: 'empty', + group: 'retry', + title: 'reasoning only, no visible output', + targets: { 'openai.stream': ['empty'], 'openai.nonstream': ['empty'], 'anthropic.stream': ['empty'], 'anthropic.nonstream': ['empty'] }, + applicable: { openai: true, anthropic: true }, + rounds: [[think('Let me consider the request before answering.'), STOP], RECOVERY_ROUND, RECOVERY_ROUND] + }, + { + id: 'bare', + group: 'retry', + title: 'prose with no completion wrapper', + targets: { 'openai.stream': ['bare'], 'openai.nonstream': ['bare'], 'anthropic.stream': [], 'anthropic.nonstream': [] }, + applicable: { openai: true, anthropic: false }, + rounds: [[answer(PLAIN_PROSE), STOP], RECOVERY_ROUND, RECOVERY_ROUND] + }, + { + id: 'invalid_control', + group: 'retry', + title: 'an unbalanced completion wrapper', + targets: { 'openai.stream': ['invalid_control'], 'openai.nonstream': ['invalid_control'], 'anthropic.stream': [], 'anthropic.nonstream': [] }, + applicable: { openai: true, anthropic: false }, + // La etiqueta abierta SIN cuerpo: con cuerpo, el texto ya salió al cliente y la guarda + // de stream invalidado (422) veta el reintento antes de que el gate lo decida — la + // celda dejaría de cubrir `invalid_control` y pasaría a cubrir esa guarda. + rounds: [[answer(''), STOP], RECOVERY_ROUND, RECOVERY_ROUND] + }, + { + id: 'good_call_with_broken_sibling', + group: 'combination', + title: 'one good parsed call beside a broken sibling', + // Asimetría deliberada (policy `toolErrorsVetoWithCalls`): OpenAI reintenta el conjunto + // (una llamada parcial es una acción silenciosamente equivocada), Anthropic entrega la + // llamada buena (bloques discretos: el cliente puede actuar con lo que llegó). + targets: { 'openai.stream': ['tool_error'], 'openai.nonstream': ['tool_error'], 'anthropic.stream': [], 'anthropic.nonstream': [] }, + applicable: { openai: true, anthropic: true }, + rounds: [[answer(`${READ_CALL}\n\n${BROKEN_SIBLING}`), STOP], RECOVERY_ROUND, RECOVERY_ROUND] + }, + { + id: 'tool_error_with_required', + group: 'combination', + title: 'a tool error together with an unsatisfied required', + // Asimetría deliberada (policy `toolErrorsBeforeRequired`): OpenAI veta por error de + // herramienta antes de mirar `required`; Anthropic mira `required` primero. + targets: { 'openai.stream': ['tool_error'], 'openai.nonstream': ['tool_error'], 'anthropic.stream': ['required_tool'], 'anthropic.nonstream': ['required_tool'] }, + applicable: { openai: true, anthropic: true }, + toolChoice: 'required', + rounds: [[answer(BROKEN_SIBLING), STOP], RECOVERY_ROUND, RECOVERY_ROUND] + }, + { + id: 'delivered_round_then_empty', + group: 'combination', + title: 'a delivered round followed by an empty round', + targets: { 'openai.stream': ['bare', 'empty'], 'openai.nonstream': ['bare', 'empty'], 'anthropic.stream': ['missing_tool'], 'anthropic.nonstream': ['missing_tool', 'empty'] }, + applicable: { openai: true, anthropic: true }, + // Ronda 1 narra sin llamar (rechazada por las tres células que deciden por ronda), ronda + // 2 sólo piensa. La tercera ronda existe para que el agotamiento no se confunda con "el + // sender se quedó seco". + // + // MEDIDO (2026-10-09): esta fila NO expone la divergencia de alcance de `empty`, que es + // lo que su comentario afirmaba antes. Mutar el juicio del stream de `visibleText` a + // `attemptVisibleText` deja el corpus BYTE-IDÉNTICO en las 64 celdas. La razón es + // estructural: cuando el texto acumulado no está vacío, la guarda de compensación + // (`if (visibleText.trim())`, anthropic.js) ya se evaluó sobre ese mismo texto acumulado + // y o bien rompió el loop o bien ya gastó el único reintento posterior a texto visible — + // así que la ronda vacía nunca llega a decidirse por el alcance de `empty`. La + // divergencia que la fila sí muestra (2 envíos en el stream contra 3 en las otras dos) + // la produce esa guarda, no la rama `empty`. + rounds: [ + [answer(UNEXECUTED_ACTION), STOP], + [think('Let me reconsider the request from scratch.'), STOP], + RECOVERY_ROUND + ] + }, + { + id: 'text_channel_cut_with_calls', + group: 'combination', + title: 'a text-channel runaway cut with calls already admitted', + targets: { 'openai.stream': [], 'openai.nonstream': [], 'anthropic.stream': [], 'anthropic.nonstream': [] }, + applicable: { openai: true, anthropic: true }, + requiresCut: true, + // La ronda repite la llamada (byte-idéntica) y después narra: la guarda corta en el + // duplicado y destruye el upstream, así que los frames siguientes no se tiran siquiera. + // El último frame es un hermano roto A PROPÓSITO: si la guarda dejara de cortar, ese + // error duro entraría al parser y la ronda se reintentaría (2 envíos) en vez de + // entregarse — el baseline distingue las dos cosas. + rounds: [ + [ + answer(READ_CALL), + answer(`\n\n${READ_CALL}`), + answer('\n\nnarration that must never be served'), + answer(`\n\n${BROKEN_SIBLING}`), + STOP + ], + RECOVERY_ROUND, + RECOVERY_ROUND + ] + } +] + +// ────────────────────────────── el runner ────────────────────────────── + +/** + * Conduce un escenario por una célula y devuelve su resultado observable. + * @param {Object} scenario - entrada de SCENARIOS + * @param {Object} surface - entrada de SURFACES + * @returns {Promise} entrada del baseline + */ +const runScenario = async (scenario, surface) => { + const sender = scriptedSender(scenario.rounds.slice(1)) + const { status, items, baseBody, upstream } = await surface.run(scenario, sender) + const hints = sender.calls.map(call => hintOf(call, baseBody)) + return { + applicable: scenario.applicable[surface.family] === true, + targets: scenario.targets[surface.id] ?? [], + status, + retried: sender.calls.length > 0, + upstreamSends: 1 + sender.calls.length, + // Frames que el handler llegó a tirar del upstream de la primera ronda. served < total + // es la huella de una guarda que abortó el intento a mitad de stream. + upstreamFrames: { served: upstream.served, total: upstream.total }, + hints, + delivered: items + } +} + +/** El corpus completo: { scenarioId: { surfaceId: entrada } }. */ +const runCorpus = async () => { + const out = {} + for (const scenario of SCENARIOS) { + out[scenario.id] = {} + for (const surface of SURFACES) { + out[scenario.id][surface.id] = await runScenario(scenario, surface) + } + } + return out +} + +/** JSON con claves ordenadas: el baseline tiene que diferenciarse limpio. */ +const sortKeys = (value) => { + if (Array.isArray(value)) return value.map(sortKeys) + if (value && typeof value === 'object') { + const out = {} + for (const key of Object.keys(value).sort()) out[key] = sortKeys(value[key]) + return out + } + return value +} + +const stableStringify = (value) => `${JSON.stringify(sortKeys(value), null, 2)}\n` + +/** + * Invariantes anti-baseline-hueco, compartidos por el grabador y el test: una sola + * implementación, porque dos copias divergen y la que decide si se graba no es la que + * decide si pasa. Una fila que declara cubrir una razón tiene que haber reintentado (si no, + * sus frames no llegan al camino que dice cubrir); una fila de aceptación no puede + * reintentar; toda fila entrega algo al cliente; una fila no aplicable no declara tokens. + * @param {Object} scenario - entrada de SCENARIOS + * @param {Object} entry - fila del baseline + * @param {string} where - etiqueta de la celda para el mensaje + * @returns {string[]} violaciones; lista vacía = fila sana + */ +const corpusViolations = (scenario, entry, where) => { + const out = [] + if (entry.delivered.length === 0) out.push(`${where}: no entrega nada al cliente`) + if (scenario.requiresCut && entry.upstreamFrames.served >= entry.upstreamFrames.total) { + out.push(`${where}: la guarda de fuga no abortó el stream (${entry.upstreamFrames.served}/${entry.upstreamFrames.total})`) + } + if (!entry.applicable) { + if (entry.targets.length > 0) out.push(`${where}: no aplicable con tokens declarados`) + return out + } + if (entry.targets.length > 0) { + if (entry.upstreamSends < 2) out.push(`${where}: declara ${entry.targets.join('+')} y no reintentó`) + // Una razón declarada que no disparó es cobertura que la fila dice tener y no tiene: la + // fila queda verde afirmando algo que nadie midió. Cada razón declarada dispara una vez, + // así que el número de hints observados tiene que coincidir con el de tokens declarados. + if (entry.hints.length !== entry.targets.length) { + out.push(`${where}: declara ${entry.targets.length} razones (${entry.targets.join('+')}) y sólo ${entry.hints.length} dispararon`) + } + } else if (entry.upstreamSends !== 1) { + out.push(`${where}: es celda de aceptación y reintentó`) + } + return out +} + +module.exports = { + SCENARIOS, + SURFACES, + runScenario, + runCorpus, + corpusViolations, + stableStringify +} diff --git a/tests/agent-turn-corpus.test.js b/tests/agent-turn-corpus.test.js new file mode 100644 index 00000000..380552f6 --- /dev/null +++ b/tests/agent-turn-corpus.test.js @@ -0,0 +1,56 @@ +'use strict' + +/** + * Test permanente del corpus de caracterización del gate de turno agéntico (issue 03). + * + * Afirma, escenario por escenario y superficie por superficie, el resultado observable + * grabado en tests/fixtures/agent-turn-corpus.baseline.json — no una comparación contra + * un archivo vivo: el baseline es el artefacto y este test lo clava. El refactor de los + * tickets 04..07 tiene que aparecer como un diff nombrado contra ese archivo, y este test + * es lo que hace que el diff no pueda pasar en silencio. + * + * El corpus en sí (frames, sender con guion, extracción del resultado) vive en + * ./agent-turn-corpus.scenarios.js, compartido con el grabador. Sin red y sin login. + */ + +const test = require('node:test') +const assert = require('node:assert/strict') + +const baseline = require('./fixtures/agent-turn-corpus.baseline.json') +const { SCENARIOS, SURFACES, runScenario, corpusViolations } = require('./agent-turn-corpus.scenarios.js') + +// account.js arranca intervalos con ref al importarse (vía controllers). +test.after(() => { + try { require('../src/utils/account.js').destroy() } catch (_) { /* nada que limpiar */ } +}) + +test('el baseline versionado cubre todas las celdas del corpus', () => { + assert.equal(baseline.corpus, 'agent-turn-corpus') + assert.equal(baseline.version, 1) + assert.deepEqual( + Object.keys(baseline.scenarios).sort(), + SCENARIOS.map(scenario => scenario.id).sort() + ) + for (const scenario of SCENARIOS) { + assert.deepEqual( + Object.keys(baseline.scenarios[scenario.id]).sort(), + SURFACES.map(surface => surface.id).sort(), + `celdas grabadas de ${scenario.id}` + ) + } +}) + +for (const scenario of SCENARIOS) { + test(`corpus de turno agéntico: ${scenario.id} — ${scenario.title}`, async () => { + for (const surface of SURFACES) { + const entry = await runScenario(scenario, surface) + const where = `${scenario.id} en ${surface.id}` + assert.deepStrictEqual(entry, baseline.scenarios[scenario.id][surface.id], where) + + // Anti-baseline-hueco: los mismos invariantes que aplica el grabador antes de escribir, + // desde la misma implementación — si divergieran, el grabador podría congelar una fila + // que el test después acepta sin mirar. + assert.deepStrictEqual(corpusViolations(scenario, entry, where), [], where) + } + }) +} diff --git a/tests/agent-turn-gate.test.js b/tests/agent-turn-gate.test.js new file mode 100644 index 00000000..a99b976f --- /dev/null +++ b/tests/agent-turn-gate.test.js @@ -0,0 +1,297 @@ +'use strict' + +/** + * Ticket 05 de `.scratch/agent-turn-gate/`: la tabla de verdad de la puerta. + * + * La puerta es una función pura de un snapshot de intento más la política de la superficie, así + * que acá no hay ruta HTTP ni config: cada fila es un snapshot, una política y el veredicto que + * el contrato exige. Cubre cada token del vocabulario, cada campo de la política (encendido y + * apagado), y las combinaciones que el corpus de caracterización (ticket 03) no tiene. + * + * El corpus sigue siendo la red de comportamiento observable de las superficies; esto es la + * red del seam: barata y exacta porque la puerta no tiene efectos. + */ + +// La puerta es una hoja (no lee config ni red), pero el logger sí mira el entorno al importarse. +process.env.DATA_SAVE_MODE = 'none' +process.env.ACCOUNTS = '' +process.env.LOG_LEVEL = 'error' +process.env.ENABLE_FILE_LOG = 'false' + +const test = require('node:test') +const assert = require('node:assert/strict') +const { + gate, + REASONS, + PROTOCOL_RECOVERY_REASONS, + TERMINAL_FINISH_REASONS, + RETRY_HINT_BUILDERS, + retryHintFor, + appendRetryHint, + FINISH_STOP, + FINISH_TOOL_CALLS +} = require('../src/utils/agent-turn-gate.js') +const { buildAgentRetryHint } = require('../src/utils/agent-turn.js') + +/** La política que declara la superficie Anthropic (stream y no-stream comparten valores). */ +const ANTHROPIC = Object.freeze({ + proseWithTools: true, + acceptBareFinal: true, + toolErrorsBeforeRequired: false, + toolErrorsVetoWithCalls: false +}) + +/** La política que el spec describe para la superficie OpenAI (todavía sin cablear). */ +const OPENAI = Object.freeze({ + proseWithTools: false, + acceptBareFinal: false, + toolErrorsBeforeRequired: true, + toolErrorsVetoWithCalls: true +}) + +/** Un intento sin nada que objetar; cada fila tuerce solo lo que necesita. */ +const snapshot = (overrides) => ({ + finishReason: 'stop', + visibleText: 'respuesta final', + controlKind: null, + toolCalls: [], + toolErrors: [], + textToolErrors: [], + nativeToolCalls: [], + interceptedToolNames: [], + thinkEvidence: false, + callsDelivered: false, + textChannelCut: false, + orphanResidue: false, + hasTools: true, + requiresToolCall: false, + ...overrides +}) + +const call = { name: 'Bash' } +const toolError = { type: 'unknown_tool', name: 'Bash' } +const NARRATED_ACTION = "I'll run the tests now." + +/** name → [snapshot overrides, policy, veredicto esperado] */ +const TABLE = [ + // --- cada token del vocabulario, por el camino más corto que lo produce --- + ['empty: sin texto y sin finish terminal', { visibleText: '' }, ANTHROPIC, + { verdict: 'retry', reason: REASONS.EMPTY }], + ['bare: la superficie no acepta prosa sin envoltorio', { controlKind: 'bare' }, OPENAI, + { verdict: 'retry', reason: REASONS.BARE }], + ['bare: sin vocabulario de control y sin texto de cierre aceptado', { visibleText: 'prosa' }, OPENAI, + { verdict: 'retry', reason: REASONS.BARE }], + ['invalid_control: el envoltorio de control no cierra', { controlKind: 'invalid_control' }, ANTHROPIC, + { verdict: 'retry', reason: REASONS.INVALID_CONTROL }], + ['required_tool: el tool_choice exigía una llamada y no hubo', { requiresToolCall: true }, ANTHROPIC, + { verdict: 'retry', reason: REASONS.REQUIRED_TOOL }], + ['tool_error: un error de herramienta sin llamada que lo acompañe', { toolErrors: [toolError] }, ANTHROPIC, + { verdict: 'retry', reason: REASONS.TOOL_ERROR }], + ['prose_with_tools: prosa junto a una llamada donde la política la veta', + { toolCalls: [call], visibleText: 'prosa' }, OPENAI, + { verdict: 'retry', reason: REASONS.PROSE_WITH_TOOLS }], + ['intercepted: frames descartados por la plataforma', { interceptedToolNames: ['Bash'] }, ANTHROPIC, + { verdict: 'retry', reason: REASONS.INTERCEPTED }], + ['malformed_protocol: residuo de protocolo en el texto', { orphanResidue: true }, ANTHROPIC, + { verdict: 'retry', reason: REASONS.MALFORMED_PROTOCOL }], + ['thought_tool_call: la llamada quedó en la razón oculta', { thinkEvidence: true }, ANTHROPIC, + { verdict: 'retry', reason: REASONS.THOUGHT_TOOL_CALL }], + ['missing_tool: texto que narra la acción y no la ejecuta', + { visibleText: NARRATED_ACTION }, ANTHROPIC, + { verdict: 'retry', reason: REASONS.MISSING_TOOL }], + + // --- un campo de política por fila, ambos valores --- + ['proseWithTools=false veta la prosa junto a llamadas', + { toolCalls: [call], visibleText: 'prosa' }, OPENAI, + { verdict: 'retry', reason: REASONS.PROSE_WITH_TOOLS }], + ['proseWithTools=true la acepta (el cliente ya tiene los bloques)', + { toolCalls: [call], visibleText: 'prosa' }, ANTHROPIC, + { verdict: 'accept', finishReason: FINISH_TOOL_CALLS }], + ['acceptBareFinal=true convierte la prosa pelada en respuesta final', + { controlKind: 'bare' }, ANTHROPIC, + { verdict: 'accept', finishReason: FINISH_STOP }], + ['acceptBareFinal=false la rechaza', { controlKind: 'bare' }, OPENAI, + { verdict: 'retry', reason: REASONS.BARE }], + ['toolErrorsBeforeRequired=false: required manda sobre el error', + { requiresToolCall: true, toolErrors: [toolError] }, ANTHROPIC, + { verdict: 'retry', reason: REASONS.REQUIRED_TOOL }], + ['toolErrorsBeforeRequired=true: el error veta antes', + { requiresToolCall: true, toolErrors: [toolError] }, OPENAI, + { verdict: 'retry', reason: REASONS.TOOL_ERROR }], + ['toolErrorsVetoWithCalls=false: la llamada parseada sobrevive al error', + { toolCalls: [call], toolErrors: [toolError] }, ANTHROPIC, + { verdict: 'accept', finishReason: FINISH_TOOL_CALLS }], + ['toolErrorsVetoWithCalls=true: el error veta aunque haya llamada', + { toolCalls: [call], toolErrors: [toolError] }, { ...ANTHROPIC, toolErrorsVetoWithCalls: true }, + { verdict: 'retry', reason: REASONS.TOOL_ERROR }], + + // --- combinaciones que el corpus no cubre --- + ['una llamada ya entregada no se retracta: gana a todo lo demás', + { callsDelivered: true, toolErrors: [toolError], interceptedToolNames: ['Bash'], visibleText: '', requiresToolCall: true }, + ANTHROPIC, { verdict: 'accept', finishReason: FINISH_TOOL_CALLS }], + ['las llamadas nativas ganan a un error de herramienta', + { nativeToolCalls: [call], toolErrors: [toolError] }, ANTHROPIC, + { verdict: 'accept', finishReason: FINISH_TOOL_CALLS }], + ['la ronda cortada con llamadas admitidas se entrega, con la prosa permitida', + { textChannelCut: true, toolCalls: [call], visibleText: 'prosa' }, ANTHROPIC, + { verdict: 'accept', finishReason: FINISH_TOOL_CALLS }], + ['ronda cortada sin llamadas: no hay nada que entregar, sigue el juicio normal', + { textChannelCut: true, visibleText: '' }, ANTHROPIC, + { verdict: 'retry', reason: REASONS.EMPTY }], + ['finish terminal apaga empty', { finishReason: 'length', visibleText: '' }, ANTHROPIC, + { verdict: 'accept', finishReason: FINISH_STOP }], + ['finish terminal apaga la evidencia (intercepted/orphan/think/narración)', + { finishReason: 'content_filter', interceptedToolNames: ['Bash'], orphanResidue: true, thinkEvidence: true, visibleText: NARRATED_ACTION }, + ANTHROPIC, { verdict: 'accept', finishReason: FINISH_STOP }], + ['finish terminal acota el veto de errores al canal de texto', + { finishReason: 'length', textToolErrors: [toolError], toolErrors: [toolError] }, ANTHROPIC, + { verdict: 'retry', reason: REASONS.TOOL_ERROR }], + ['finish terminal: un error solo nativo no veta', + { finishReason: 'length', textToolErrors: [], toolErrors: [{ type: 'invalid_arguments', name: 'Bash' }] }, ANTHROPIC, + { verdict: 'accept', finishReason: FINISH_STOP }], + ['sin herramientas no hay required ni evidencia que valga', + { hasTools: false, requiresToolCall: true, interceptedToolNames: ['Bash'], thinkEvidence: true }, + ANTHROPIC, { verdict: 'accept', finishReason: FINISH_STOP }], + ['required_tool no aplica si la llamada llegó', + { requiresToolCall: true, toolCalls: [call] }, ANTHROPIC, + { verdict: 'accept', finishReason: FINISH_TOOL_CALLS }], + ['controlKind final con texto: respuesta final', { controlKind: 'final' }, OPENAI, + { verdict: 'accept', finishReason: FINISH_STOP }], + ['controlKind final sin texto: empty', { controlKind: 'final', visibleText: '' }, OPENAI, + { verdict: 'retry', reason: REASONS.EMPTY }], + // El cupo de recuperación gastado apaga el REINTENTO por residuo, no la supresión de la + // entrega: el código viejo medía el residuo sin condición para decidir si el texto viajaba + // y gateaba sólo el reintento. Sin estas dos filas, apagar el campo entero —que es lo que + // la revisión de spec encontró— volvía a colar bytes de protocolo al cliente. + // Lo que se afirma acá es la RAZÓN, no el veredicto: con el cupo gastado la ronda cae a + // `bare` (sin envoltorio) en vez de reintentarse por `malformed_protocol`, que es lo que + // hacía el código viejo al saltarse el chequeo. + ['residuo con el cupo gastado: no reintenta por malformed_protocol', + { orphanResidue: true, protocolRecoverySpent: true }, OPENAI, + { verdict: 'retry', reason: REASONS.BARE }], + ['residuo con el cupo gastado sobre una aceptación nativa: el texto igual se suprime', + { orphanResidue: true, protocolRecoverySpent: true, nativeToolCalls: [call] }, OPENAI, + { verdict: 'accept', finishReason: FINISH_TOOL_CALLS, suppressVisibleText: true }], + ['residuo sin cupo gastado: sí reintenta', { orphanResidue: true }, OPENAI, + { verdict: 'retry', reason: REASONS.MALFORMED_PROTOCOL }], + ['controlKind blocked con texto: también es una respuesta final', + { controlKind: 'blocked' }, OPENAI, { verdict: 'accept', finishReason: FINISH_STOP }], + ['controlKind blocked sin texto: empty, igual que final', + { controlKind: 'blocked', visibleText: '' }, OPENAI, + { verdict: 'retry', reason: REASONS.EMPTY }], + ['controlKind empty: la ronda vacía se reintenta', { controlKind: 'empty' }, OPENAI, + { verdict: 'retry', reason: REASONS.EMPTY }], + ['missing_tool es de las superficies sin vocabulario de control', + { controlKind: 'final', visibleText: NARRATED_ACTION }, OPENAI, + { verdict: 'accept', finishReason: FINISH_STOP }] +] + +const run = (overrides, policy) => gate(snapshot(overrides), policy) + +for (const [name, overrides, policy, want] of TABLE) { + test(name, () => { + const verdict = run(overrides, policy) + assert.equal(verdict.verdict, want.verdict) + if (want.reason) assert.equal(verdict.reason, want.reason) + if (want.finishReason) assert.equal(verdict.finishReason, want.finishReason) + }) +} + +test('la escalera de evidencia: intercepted > malformed > thought > missing', () => { + const all = { + interceptedToolNames: ['Bash'], + orphanResidue: true, + thinkEvidence: true, + visibleText: NARRATED_ACTION + } + assert.equal(run(all, ANTHROPIC).reason, REASONS.INTERCEPTED) + assert.equal(run({ ...all, interceptedToolNames: [] }, ANTHROPIC).reason, REASONS.MALFORMED_PROTOCOL) + assert.equal(run({ ...all, interceptedToolNames: [], orphanResidue: false }, ANTHROPIC).reason, REASONS.THOUGHT_TOOL_CALL) + assert.equal(run({ ...all, interceptedToolNames: [], orphanResidue: false, thinkEvidence: false }, ANTHROPIC).reason, REASONS.MISSING_TOOL) +}) + +test('la ronda cortada con llamadas suprime el texto que la política no permite', () => { + const cut = { textChannelCut: true, toolCalls: [call], visibleText: 'prosa' } + assert.equal(run(cut, ANTHROPIC).suppressVisibleText, false) + assert.equal(run(cut, OPENAI).suppressVisibleText, true, 'prosaWithTools=false la veta al entregar') + assert.equal(run({ ...cut, toolErrors: [toolError] }, ANTHROPIC).suppressVisibleText, true) + assert.equal(run({ textChannelCut: true, toolCalls: [call], orphanResidue: true }, ANTHROPIC).suppressVisibleText, true) +}) + +test('las llamadas nativas suprimen el texto solo con pruebas de que trae basura', () => { + assert.equal(run({ nativeToolCalls: [call] }, ANTHROPIC).suppressVisibleText, false) + assert.equal(run({ nativeToolCalls: [call], textToolErrors: [toolError] }, ANTHROPIC).suppressVisibleText, true) + assert.equal(run({ nativeToolCalls: [call], orphanResidue: true }, ANTHROPIC).suppressVisibleText, true) +}) + +test('forma del veredicto: accept no lleva razón, retry no lleva finish reason', () => { + for (const [name, overrides, policy] of TABLE) { + const verdict = run(overrides, policy) + if (verdict.verdict === 'accept') { + assert.equal(verdict.reason, null, name) + assert.ok(verdict.finishReason, name) + } else { + assert.equal(verdict.finishReason, null, name) + assert.ok(Object.values(REASONS).includes(verdict.reason), name) + assert.equal(verdict.suppressVisibleText, false, name) + } + } +}) + +test('max_tokens es terminal igual que length', () => { + assert.ok(TERMINAL_FINISH_REASONS.has('max_tokens')) + assert.equal(run({ finishReason: 'max_tokens', visibleText: '' }, ANTHROPIC).verdict, 'accept') + assert.notEqual(run({ finishReason: 'max_tokens', visibleText: '' }, OPENAI).reason, REASONS.EMPTY) +}) + +test('cada token del vocabulario tiene constructor de hint y produce texto', () => { + const hintSnapshot = snapshot({ toolErrors: [toolError], textToolErrors: [toolError], interceptedToolNames: ['Bash'], thinkEvidence: true }) + for (const token of Object.values(REASONS)) { + assert.ok(RETRY_HINT_BUILDERS[token], `sin constructor: ${token}`) + const hint = retryHintFor(token, hintSnapshot, { toolChoice: 'required', allowedToolNames: ['Bash'] }) + assert.equal(typeof hint, 'string', token) + assert.ok(hint.trim().length > 0, token) + } +}) + +test('el cupo de recuperación de protocolo es exactamente intercepted/malformed/thought', () => { + assert.deepEqual([...PROTOCOL_RECOVERY_REASONS].sort(), + [REASONS.INTERCEPTED, REASONS.MALFORMED_PROTOCOL, REASONS.THOUGHT_TOOL_CALL].sort()) + assert.ok(!PROTOCOL_RECOVERY_REASONS.has(REASONS.MISSING_TOOL)) + assert.ok(!PROTOCOL_RECOVERY_REASONS.has(REASONS.TOOL_ERROR)) + assert.ok(!PROTOCOL_RECOVERY_REASONS.has(REASONS.EMPTY)) +}) + +test('el hint de required_tool lleva el hecho de la intercepción que lo tapa', () => { + const s = snapshot({ requiresToolCall: true, interceptedToolNames: ['Bash'] }) + const base = RETRY_HINT_BUILDERS[REASONS.REQUIRED_TOOL](s, { toolChoice: 'required' }) + assert.equal(retryHintFor(REASONS.REQUIRED_TOOL, s, { toolChoice: 'required' }), + `${base}\n${buildAgentRetryHint('intercepted')}`) +}) + +test('el hint de tool_error lleva el hecho del think leak que lo tapa', () => { + const s = snapshot({ toolErrors: [toolError], thinkEvidence: true }) + const base = RETRY_HINT_BUILDERS[REASONS.TOOL_ERROR](s, { allowedToolNames: ['Bash'] }) + assert.equal(retryHintFor(REASONS.TOOL_ERROR, s, { allowedToolNames: ['Bash'] }), + `${base}\n${buildAgentRetryHint('thought_tool_call')}`) +}) + +test('appendRetryHint: el encabezado va solo en los brazos que agregan a un texto', () => { + const header = '# Tool-call retry' + const stringBody = { messages: [{ role: 'user', content: 'contexto' }] } + assert.equal(appendRetryHint(stringBody, 'HINT', { header }).messages[0].content, `contexto\n\n${header}\nHINT`) + assert.equal(appendRetryHint(stringBody, 'HINT').messages[0].content, 'contexto\n\nHINT') + const partsBody = { messages: [{ role: 'user', content: [{ type: 'text', text: 'contexto' }] }] } + assert.equal(appendRetryHint(partsBody, 'HINT', { header }).messages[0].content[0].text, `contexto\n\n${header}\nHINT`) + const noTextPart = { messages: [{ role: 'user', content: [{ type: 'image', source: {} }] }] } + assert.deepEqual(appendRetryHint(noTextPart, 'HINT', { header }).messages[0].content[0], { type: 'text', text: 'HINT' }) + assert.deepEqual(appendRetryHint({ messages: [] }, 'HINT', { header }).messages[0], { role: 'user', content: 'HINT' }) +}) + +test('appendRetryHint no muta el cuerpo original (se reusa entre reintentos)', () => { + const body = { messages: [{ role: 'user', content: 'contexto' }] } + const out = appendRetryHint(body, 'HINT', { header: '# Tool-call retry' }) + assert.equal(body.messages[0].content, 'contexto') + assert.notEqual(out, body) + assert.equal(out.messages[0].content, 'contexto\n\n# Tool-call retry\nHINT') +}) diff --git a/tests/anthropic-failure-status.test.js b/tests/anthropic-failure-status.test.js new file mode 100644 index 00000000..826c68f5 --- /dev/null +++ b/tests/anthropic-failure-status.test.js @@ -0,0 +1,97 @@ +// Gemelo de tests/openai-failure-status.test.js en la superficie Anthropic: la salida +// pre-respuesta de /v1/messages contestaba 500 `api_error` para todo. Ahora el veredicto que +// viaja con el fallo se traduce al vocabulario de Anthropic. Se conduce el handler con un +// doble de res y se afirma lo que recibe el cliente. +const test = require('node:test') +const assert = require('node:assert/strict') + +process.env.API_KEY = 'anthropic-failure-status-test-key' +process.env.DATA_SAVE_MODE = 'none' +process.env.ACCOUNTS = '' +process.env.ENABLE_CLI = 'false' +process.env.ENABLE_FILE_LOG = 'false' +process.env.PROXY_URL = '' + +const modelsMap = require('../src/models/models-map.js') +modelsMap.getLatestModels = async () => { throw new Error('offline test: no model fetch') } + +// El controlador captura sendChatRequest por destructuring en su primer require. +const requestModule = require('../src/utils/request.js') +let upstreamResult = { status: false } +requestModule.sendChatRequest = async () => upstreamResult + +const { handleAnthropicMessages } = require('../src/controllers/anthropic.js') + +test.after(() => { require('../src/utils/account.js').destroy() }) + +const jsonRes = () => ({ + statusCode: 200, + body: null, + headers: {}, + headersSent: false, + writableEnded: false, + set(h, v) { if (typeof h === 'string') this.headers[h] = v; else Object.assign(this.headers, h); return this; }, + status(code) { this.statusCode = code; return this; }, + json(payload) { this.body = payload; this.headersSent = true; this.writableEnded = true; return this; }, + write(chunk) { this.headersSent = true; this.output = (this.output || '') + String(chunk); return true; }, + end(chunk = '') { if (chunk) this.output = (this.output || '') + String(chunk); this.writableEnded = true; } +}) + +const failure = (overrides) => ({ + status: false, + response: null, + failure: { rateLimited: false, overloaded: false, status: 502, retryAfter: null, ...overrides } +}) + +const drive = async (result) => { + upstreamResult = result + const res = jsonRes() + await handleAnthropicMessages({ + body: { model: 'qwen3-max', max_tokens: 64, stream: false, messages: [{ role: 'user', content: 'hola' }] } + }, res) + return res +} + +test('la cuota agotada sale 429 rate_limit_error, no un 500 api_error', async () => { + const res = await drive(failure({ rateLimited: true, status: 429, retryAfter: 3600 })) + + assert.equal(res.statusCode, 429) + assert.equal(res.body.type, 'error') + assert.equal(res.body.error.type, 'rate_limit_error') + assert.equal(res.headers['Retry-After'], '3600') +}) + +test('sin espera del upstream no se inventa Retry-After', async () => { + const res = await drive(failure({ rateLimited: true, status: 429 })) + + assert.equal(res.statusCode, 429) + assert.equal(res.headers['Retry-After'], undefined) +}) + +test('un upstream sobrecargado sale 529 overloaded_error', async () => { + const res = await drive(failure({ overloaded: true, status: 529, retryAfter: 10 })) + + assert.equal(res.statusCode, 529) + assert.equal(res.body.error.type, 'overloaded_error') + assert.equal(res.headers['Retry-After'], '10') +}) + +test('el transporte agotado sale 503', async () => { + const res = await drive(failure({ status: 503 })) + + assert.equal(res.statusCode, 503) + assert.equal(res.body.error.type, 'api_error') +}) + +test('un no-200 opaco del upstream sale 502, no 500', async () => { + const res = await drive(failure({ status: 502 })) + + assert.equal(res.statusCode, 502) + assert.equal(res.body.error.type, 'api_error') +}) + +test('la razon concreta del modulo de request sigue llegando al cliente', async () => { + const res = await drive({ ...failure({ status: 503 }), message: 'no hay cuentas configuradas' }) + + assert.equal(res.body.error.message, 'no hay cuentas configuradas') +}) diff --git a/tests/anthropic-interception-retry.test.js b/tests/anthropic-interception-retry.test.js index 811b9fb7..e189a7c8 100644 --- a/tests/anthropic-interception-retry.test.js +++ b/tests/anthropic-interception-retry.test.js @@ -328,9 +328,12 @@ describe('interception observability (finding 3)', () => { hint = JSON.stringify(sender.calls[0]); }); + // El token de la razon es el unificado del vocabulario (`required_tool`, ticket 05): + // antes decia `required`. Lo que esta prueba cuida — que el log diga los nombres + // descartados aunque otra razon los tape — no cambia. assert.ok( - warns.some(line => /required; dropped: read_file/.test(line)), - `expected a "required; dropped: read_file" warn, got:\n${warns.join('\n')}` + warns.some(line => /required_tool; dropped: read_file/.test(line)), + `expected a "required_tool; dropped: read_file" warn, got:\n${warns.join('\n')}` ); // finding 4 para la razon required: el hint lleva el dato clave ademas del suyo. assert.match(hint, /You did not call any tool/); diff --git a/tests/anthropic-nonstream-retry-filter.test.js b/tests/anthropic-nonstream-retry-filter.test.js new file mode 100644 index 00000000..879b12f2 --- /dev/null +++ b/tests/anthropic-nonstream-retry-filter.test.js @@ -0,0 +1,142 @@ +/** + * Ticket 02 (`.scratch/agent-turn-gate/issues/02-filtro-de-frames-no-stream.md`). + * + * `createUpstreamResponseFilter` latcha el `response_id` que acepta, y el loop + * no-stream (C) lo construye UNA sola vez por petición: su bloque de reintento + * reconstruye el normalizador, los acumuladores y el estado por ronda, pero no el + * filtro. El plan de unificación de loops (lohari, 2026-08-31) afirmó que por eso + * un reintento con otro `response_id` no contribuye nada. + * + * MEDICIÓN (2026-10-09): la afirmación es cierta sobre el mecanismo y falsa sobre + * el disparador. Un reintento que abre con `response.created` —que es como abre + * TODA generación del upstream, ver los frames capturados en `sse.test.js:130` y + * `chat-challenge.test.js:56`— hace que el filtro RE-latchee al id nuevo, así que + * sus frames entran con normalidad. La forma que sí pierde los frames es un + * reintento SIN `response.created` y con otro id: la rama de protocolo viejo del + * filtro conserva el primer id y descarta todo lo demás. Esa forma se midió y se + * descartó por inalcanzable; no se toca el loop. + * + * El test de abajo es la red de regresión sobre el re-latch: si alguien cambia el + * filtro para latchar de forma permanente, o deja de reconstruir la ventana de + * ids, esto se pone rojo. + */ + +const test = require('node:test') +const assert = require('node:assert/strict') +const { Readable } = require('node:stream') +const { handleAnthropicNonStream } = require('../src/controllers/anthropic.js') + +test.after(() => { + require('../src/utils/account.js').destroy() +}) + +// --------------------------------------------------------------------------- +// Harness (mismas formas que anthropic-native-parity.test.js) +// --------------------------------------------------------------------------- + +const createMockJsonResponse = () => ({ + statusCode: 200, + body: null, + headers: {}, + set(headers) { Object.assign(this.headers, headers); return this; }, + status(code) { this.statusCode = code; return this; }, + json(payload) { this.body = payload; return this; } +}) + +const frame = (payload) => `data: ${JSON.stringify(payload)}\n\n` + +const createdFrame = (id, index = '0') => frame({ + 'response.created': { chat_id: 'c1', parent_id: 'p1', response_id: id, response_index: index } +}) + +const textFrame = (id, text) => frame({ + choices: [{ delta: { role: 'assistant', content: text, phase: 'answer', status: 'typing' }, finish_reason: null }], + response_id: id +}) + +const nativeCallFrame = (id, name, snapshot) => frame({ + choices: [{ + delta: { + role: 'assistant', + content: '', + phase: 'answer', + status: 'typing', + function_call: { name, arguments: snapshot }, + extra: { display_position: 'answer' } + }, + finish_reason: null + }], + response_id: id +}) + +const notExistsFrame = (id, name) => frame({ + choices: [{ + delta: { role: 'function', content: `Tool ${name} does not exists.`, phase: 'answer', status: 'typing', name }, + finish_reason: null + }], + response_id: id +}) + +const terminator = (id, finishReason) => frame({ + choices: [{ delta: {}, finish_reason: finishReason }], + response_id: id +}) + 'data: [DONE]\n\n' + +const BASH_ARGS = '{"command": "git status"}' +const BASH_SNAPSHOTS = ['', '{"command": ', '{"command": "git status"', BASH_ARGS, BASH_ARGS] + +/** Primer intento: prosa sin llamada → `required` sin cumplir → un reintento. */ +const firstAttempt = (id) => () => Readable.from([ + createdFrame(id), + textFrame(id, 'Voy a revisar el repositorio.'), + terminator(id, 'stop') +]) + +/** Reintento con la llamada nativa completa, bajo el id que se le pase. */ +const retryAttempt = (id) => () => Readable.from([ + createdFrame(id), + ...BASH_SNAPSHOTS.map(snapshot => nativeCallFrame(id, 'Bash', snapshot)), + notExistsFrame(id, 'Bash'), + terminator(id, 'stop') +]) + +const scriptedSender = (...turns) => { + const queue = [...turns] + const fn = async (body) => { + fn.calls.push(body) + const next = queue.shift() + return next ? { status: true, response: next() } : { status: false } + } + fn.calls = [] + return fn +} + +const baseCtx = (sendRequest) => ({ + message_id: 'msg_retry_filter', + model: 'qwen-test', + hasTools: true, + toolChoice: 'required', + allowedToolNames: ['Bash'], + requestBody: { messages: [] }, + sendRequest +}) + +const runNonStream = (upstream, sendRequest) => { + const res = createMockJsonResponse() + return handleAnthropicNonStream(res, baseCtx(sendRequest), upstream()).then(() => res) +} + +const toolUseNames = (body) => (body?.content || []) + .filter(block => block.type === 'tool_use') + .map(block => block.name) + +// --------------------------------------------------------------------------- + +test('non-stream C: un reintento con OTRO response_id recupera la llamada (el filtro re-latchea con response.created)', async () => { + const sender = scriptedSender(retryAttempt('r2')) + const res = await runNonStream(firstAttempt('r1'), sender) + + assert.equal(sender.calls.length, 1, 'tool_choice=required sin llamada dispara exactamente un reintento') + assert.deepEqual(toolUseNames(res.body), ['Bash'], + 'el reintento abre con response.created: el filtro re-latchea y sus frames no se descartan') +}) diff --git a/tests/anthropic-tool-error-hints.test.js b/tests/anthropic-tool-error-hints.test.js index 275da2a8..42259d86 100644 --- a/tests/anthropic-tool-error-hints.test.js +++ b/tests/anthropic-tool-error-hints.test.js @@ -1,7 +1,12 @@ const test = require('node:test') const assert = require('node:assert/strict') -const { describeToolErrors, buildToolErrorRetryHint } = require('../src/controllers/anthropic.js') +// `describeToolErrors` es del controlador; `buildToolErrorRetryHint` vive en la puerta desde +// el ticket 05 (el re-export del controlador murió con su último llamador local, ticket 06). +// Se importa de donde honestamente vive: un alias en el controlador volvería a poner dos +// nombres para una sola función. +const { describeToolErrors } = require('../src/controllers/anthropic.js') +const { buildToolErrorRetryHint } = require('../src/utils/agent-turn-gate.js') // Los tipos que produce createNativeToolCallAccumulator en modo snapshot. Antes de D1 el // resumen los ignoraba y el log decia "unspecified" para una ronda entera de errores nativos. diff --git a/tests/anthropic-turn-gate-parity.test.js b/tests/anthropic-turn-gate-parity.test.js new file mode 100644 index 00000000..68966a44 --- /dev/null +++ b/tests/anthropic-turn-gate-parity.test.js @@ -0,0 +1,74 @@ +'use strict' + +/** + * Paridad de decisión entre las dos rutas Anthropic, en el seam de la ruta HTTP (issue 06 + * de `.scratch/agent-turn-gate/`). + * + * El corpus (`tests/agent-turn-corpus.test.js`) clava cada celda contra el baseline grabado. + * Acá se afirma la relación ENTRE las dos celdas Anthropic: conducidos los mismos escenarios + * de decisión por `/v1/messages` con y sin streaming, las dos rutas tienen que llegar a la + * misma decisión de turno — mismo status, mismo "reintentó o no", y el mismo hint viajando + * al modelo, en el mismo orden. Es lo que se ve de la puerta única desde afuera: si un + * refactor vuelve a separar los dos loops, esto rompe acá aunque el baseline se re-grabe. + * + * Las desviaciones están medidas y listadas (abajo): son diferencias de LOOP, no de puerta. + * Una desviación que no esté en la lista falla; una entrada de la lista que ya no desvíe + * también, para que la lista no envejezca tapando paridad. + * + * El módulo del corpus se importa PRIMERO: fija los pines de entorno (AGENT_TURN_MAX_ATTEMPTS + * y compañía) antes de que cualquier require arrastre config/index.js. + */ + +const { SCENARIOS, SURFACES, runScenario } = require('./agent-turn-corpus.scenarios.js') + +const test = require('node:test') +const assert = require('node:assert/strict') + +const STREAM = SURFACES.find(surface => surface.id === 'anthropic.stream') +const NON_STREAM = SURFACES.find(surface => surface.id === 'anthropic.nonstream') + +/** + * Desviaciones medidas entre las dos rutas, con su causa. Estar acá no dispensa de la + * aserción: la forma esperada sigue siendo "el no-stream decide al menos cada ronda que + * decidió el stream, con el mismo hint" y se afirma. + */ +const DESVIACIONES = new Map([ + ['delivered_round_then_empty', + 'el loop streaming ya gastó su único reintento posterior a texto visible ' + + '(retriedAfterVisibleText: la prosa de la ronda 1 salió al cliente) y no puede pedir la ' + + 'ronda 3; el no-stream no entrega nada hasta el final, así que decide también la ronda ' + + 'vacía y la reintenta. Cupo del loop, no de la puerta.'] +]) + +// account.js arranca intervalos con ref al importarse (vía controllers). +test.after(() => { + try { require('../src/utils/account.js').destroy() } catch (_) { /* nada que limpiar */ } +}) + +for (const scenario of SCENARIOS) { + test(`paridad de la puerta Anthropic: ${scenario.id} — ${scenario.title}`, async () => { + const stream = await runScenario(scenario, STREAM) + const nonStream = await runScenario(scenario, NON_STREAM) + const where = scenario.id + + assert.equal(nonStream.status, stream.status, `${where}: el status del cable difiere`) + assert.equal(nonStream.retried, stream.retried, `${where}: la decisión de reintentar difiere`) + // La guarda de fuga corta el mismo frame en las dos rutas: el punto de aborto del intento + // es parte de la misma decisión de turno. + assert.deepEqual(nonStream.upstreamFrames, stream.upstreamFrames, + `${where}: las dos rutas tiraron distinta cantidad de frames del upstream`) + + const desviacion = DESVIACIONES.get(scenario.id) + if (!desviacion) { + assert.equal(nonStream.upstreamSends, stream.upstreamSends, + `${where}: mismo veredicto tiene que dar los mismos envíos al upstream`) + assert.deepEqual(nonStream.hints, stream.hints, `${where}: los hints que viajaron al modelo difieren`) + return + } + + assert.ok(nonStream.upstreamSends > stream.upstreamSends, + `${where}: declarada como desviación y ya no desvía — sacarla de DESVIACIONES (${desviacion})`) + assert.deepEqual(nonStream.hints.slice(0, stream.hints.length), stream.hints, + `${where}: la desviación tiene que EXTENDER la decisión del stream, no cambiarla`) + }) +} diff --git a/tests/expected-counts.json b/tests/expected-counts.json index 710c2e8e..a1e0d1fb 100644 --- a/tests/expected-counts.json +++ b/tests/expected-counts.json @@ -1,6 +1,6 @@ { - "tests": 1229, + "tests": 1337, "suites": 137, "note": "Authoritative count. Verify with the per-file sum in AGENTS.md (\"The test gate\"). The -a on that grep is load-bearing: tool-prompt.test.js emits bytes that make grep call the stream binary, and without -a its whole summary line — 133 tests — is silently dropped from the sum.", - "updated": "2026-09-27" + "updated": "2026-10-09" } diff --git a/tests/fixtures/agent-turn-corpus.baseline.json b/tests/fixtures/agent-turn-corpus.baseline.json new file mode 100644 index 00000000..86aa4d1e --- /dev/null +++ b/tests/fixtures/agent-turn-corpus.baseline.json @@ -0,0 +1,2481 @@ +{ + "corpus": "agent-turn-corpus", + "scenarios": { + "accept_final_answer": { + "anthropic.nonstream": { + "applicable": true, + "delivered": [ + { + "kind": "content", + "text": "All requested work is complete." + }, + { + "kind": "stop_reason", + "stop_reason": "end_turn" + } + ], + "hints": [], + "retried": false, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 1 + }, + "anthropic.stream": { + "applicable": true, + "delivered": [ + { + "kind": "message_start" + }, + { + "index": 0, + "kind": "text_block_start" + }, + { + "kind": "content", + "text": "All requested work is complete." + }, + { + "index": 0, + "kind": "block_stop" + }, + { + "kind": "stop_reason", + "stop_reason": "end_turn" + }, + { + "kind": "message_stop" + } + ], + "hints": [], + "retried": false, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 1 + }, + "openai.nonstream": { + "applicable": true, + "delivered": [ + { + "kind": "content", + "text": "All requested work is complete." + }, + { + "kind": "finish", + "reason": "stop" + }, + { + "kind": "usage" + } + ], + "hints": [], + "retried": false, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 1 + }, + "openai.stream": { + "applicable": true, + "delivered": [ + { + "kind": "role", + "role": "assistant" + }, + { + "kind": "content", + "text": "All requested work is complete." + }, + { + "kind": "finish", + "reason": "stop" + }, + { + "kind": "usage" + }, + { + "kind": "done" + } + ], + "hints": [], + "retried": false, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 1 + } + }, + "accept_tool_call": { + "anthropic.nonstream": { + "applicable": true, + "delivered": [ + { + "id": "toolu_", + "input": { + "file_path": "a.txt" + }, + "kind": "tool_use", + "name": "Read" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + } + ], + "hints": [], + "retried": false, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 1 + }, + "anthropic.stream": { + "applicable": true, + "delivered": [ + { + "kind": "message_start" + }, + { + "id": "toolu_", + "index": 0, + "kind": "tool_use_start", + "name": "Read" + }, + { + "index": 0, + "kind": "tool_use_args", + "partial_json": "{\"file_path\":\"a.txt\"}" + }, + { + "index": 0, + "kind": "block_stop" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + }, + { + "kind": "message_stop" + } + ], + "hints": [], + "retried": false, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 1 + }, + "openai.nonstream": { + "applicable": true, + "delivered": [ + { + "arguments": "{\"file_path\":\"a.txt\"}", + "id": "call_", + "index": null, + "kind": "tool_call", + "name": "Read" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + } + ], + "hints": [], + "retried": false, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 1 + }, + "openai.stream": { + "applicable": true, + "delivered": [ + { + "kind": "role", + "role": "assistant" + }, + { + "id": "call_", + "index": 0, + "kind": "tool_call", + "name": "Read" + }, + { + "arguments": "{\"file_path\":\"a.txt\"}", + "index": 0, + "kind": "tool_call_args" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + }, + { + "kind": "done" + } + ], + "hints": [], + "retried": false, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 1 + } + }, + "bare": { + "anthropic.nonstream": { + "applicable": false, + "delivered": [ + { + "kind": "content", + "text": "The answer is 42, and no tool is needed for it." + }, + { + "kind": "stop_reason", + "stop_reason": "end_turn" + } + ], + "hints": [], + "retried": false, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 1 + }, + "anthropic.stream": { + "applicable": false, + "delivered": [ + { + "kind": "message_start" + }, + { + "index": 0, + "kind": "text_block_start" + }, + { + "kind": "content", + "text": "The answer is 42, and no tool is needed for it." + }, + { + "index": 0, + "kind": "block_stop" + }, + { + "kind": "stop_reason", + "stop_reason": "end_turn" + }, + { + "kind": "message_stop" + } + ], + "hints": [], + "retried": false, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 1 + }, + "openai.nonstream": { + "applicable": true, + "delivered": [ + { + "arguments": "{\"file_path\":\"a.txt\"}", + "id": "call_", + "index": null, + "kind": "tool_call", + "name": "Read" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt returned bare prose without declaring a verified final result or emitting the next tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "bare" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "openai.stream": { + "applicable": true, + "delivered": [ + { + "kind": "role", + "role": "assistant" + }, + { + "id": "call_", + "index": 0, + "kind": "tool_call", + "name": "Read" + }, + { + "arguments": "{\"file_path\":\"a.txt\"}", + "index": 0, + "kind": "tool_call_args" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + }, + { + "kind": "done" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt returned bare prose without declaring a verified final result or emitting the next tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "bare" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + } + }, + "delivered_round_then_empty": { + "anthropic.nonstream": { + "applicable": true, + "delivered": [ + { + "kind": "thinking", + "text": "Let me reconsider the request from scratch." + }, + { + "id": "toolu_", + "input": { + "file_path": "a.txt" + }, + "kind": "tool_use", + "name": "Read" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + } + ], + "hints": [ + "# Tool-call retry\nYour previous reply described an action but did not execute any tool call. Perform that action now by emitting the real `[TOOL CALL]` block immediately with no preamble. Do not describe the action again or claim completion without a tool result.", + "# Tool-call retry\nYour previous reply produced no visible final answer or executable tool call. Continue the Agent task now. If any action remains, emit the required `[TOOL CALL]` block immediately with no preamble. Only give a normal final answer when the task is actually complete; do not repeat hidden reasoning." + ], + "retried": true, + "status": 200, + "targets": [ + "missing_tool", + "empty" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 3 + }, + "anthropic.stream": { + "applicable": true, + "delivered": [ + { + "kind": "message_start" + }, + { + "index": 0, + "kind": "text_block_start" + }, + { + "kind": "content", + "text": "I will read the file now and then summarise what it contains." + }, + { + "index": 0, + "kind": "block_stop" + }, + { + "index": 1, + "kind": "thinking_block_start" + }, + { + "kind": "thinking", + "text": "Let me reconsider the request from scratch." + }, + { + "index": 1, + "kind": "signature" + }, + { + "index": 1, + "kind": "block_stop" + }, + { + "kind": "stop_reason", + "stop_reason": "end_turn" + }, + { + "kind": "message_stop" + } + ], + "hints": [ + "# Tool-call retry\nYour previous reply described an action but did not execute any tool call. Perform that action now by emitting the real `[TOOL CALL]` block immediately with no preamble. Do not describe the action again or claim completion without a tool result." + ], + "retried": true, + "status": 200, + "targets": [ + "missing_tool" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "openai.nonstream": { + "applicable": true, + "delivered": [ + { + "arguments": "{\"file_path\":\"a.txt\"}", + "id": "call_", + "index": null, + "kind": "tool_call", + "name": "Read" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt returned bare prose without declaring a verified final result or emitting the next tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose.", + "# Agent turn recovery\nThe previous attempt ended without a visible answer or executable tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "bare", + "empty" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 3 + }, + "openai.stream": { + "applicable": true, + "delivered": [ + { + "kind": "role", + "role": "assistant" + }, + { + "kind": "reasoning", + "text": "Let me reconsider the request from scratch." + }, + { + "id": "call_", + "index": 0, + "kind": "tool_call", + "name": "Read" + }, + { + "arguments": "{\"file_path\":\"a.txt\"}", + "index": 0, + "kind": "tool_call_args" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + }, + { + "kind": "done" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt returned bare prose without declaring a verified final result or emitting the next tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose.", + "# Agent turn recovery\nThe previous attempt ended without a visible answer or executable tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "bare", + "empty" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 3 + } + }, + "empty": { + "anthropic.nonstream": { + "applicable": true, + "delivered": [ + { + "kind": "thinking", + "text": "Let me consider the request before answering." + }, + { + "id": "toolu_", + "input": { + "file_path": "a.txt" + }, + "kind": "tool_use", + "name": "Read" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + } + ], + "hints": [ + "# Tool-call retry\nYour previous reply produced no visible final answer or executable tool call. Continue the Agent task now. If any action remains, emit the required `[TOOL CALL]` block immediately with no preamble. Only give a normal final answer when the task is actually complete; do not repeat hidden reasoning." + ], + "retried": true, + "status": 200, + "targets": [ + "empty" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "anthropic.stream": { + "applicable": true, + "delivered": [ + { + "kind": "message_start" + }, + { + "index": 0, + "kind": "thinking_block_start" + }, + { + "kind": "thinking", + "text": "Let me consider the request before answering." + }, + { + "index": 0, + "kind": "signature" + }, + { + "index": 0, + "kind": "block_stop" + }, + { + "id": "toolu_", + "index": 1, + "kind": "tool_use_start", + "name": "Read" + }, + { + "index": 1, + "kind": "tool_use_args", + "partial_json": "{\"file_path\":\"a.txt\"}" + }, + { + "index": 1, + "kind": "block_stop" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + }, + { + "kind": "message_stop" + } + ], + "hints": [ + "# Tool-call retry\nYour previous reply produced no visible final answer or executable tool call. Continue the Agent task now. If any action remains, emit the required `[TOOL CALL]` block immediately with no preamble. Only give a normal final answer when the task is actually complete; do not repeat hidden reasoning." + ], + "retried": true, + "status": 200, + "targets": [ + "empty" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "openai.nonstream": { + "applicable": true, + "delivered": [ + { + "arguments": "{\"file_path\":\"a.txt\"}", + "id": "call_", + "index": null, + "kind": "tool_call", + "name": "Read" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt ended without a visible answer or executable tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "empty" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "openai.stream": { + "applicable": true, + "delivered": [ + { + "kind": "role", + "role": "assistant" + }, + { + "kind": "reasoning", + "text": "Let me consider the request before answering." + }, + { + "id": "call_", + "index": 0, + "kind": "tool_call", + "name": "Read" + }, + { + "arguments": "{\"file_path\":\"a.txt\"}", + "index": 0, + "kind": "tool_call_args" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + }, + { + "kind": "done" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt ended without a visible answer or executable tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "empty" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + } + }, + "good_call_with_broken_sibling": { + "anthropic.nonstream": { + "applicable": true, + "delivered": [ + { + "id": "toolu_", + "input": { + "file_path": "a.txt" + }, + "kind": "tool_use", + "name": "Read" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + } + ], + "hints": [], + "retried": false, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 1 + }, + "anthropic.stream": { + "applicable": true, + "delivered": [ + { + "kind": "message_start" + }, + { + "index": 0, + "kind": "text_block_start" + }, + { + "kind": "content", + "text": "\n\n" + }, + { + "index": 0, + "kind": "block_stop" + }, + { + "id": "toolu_", + "index": 1, + "kind": "tool_use_start", + "name": "Read" + }, + { + "index": 1, + "kind": "tool_use_args", + "partial_json": "{\"file_path\":\"a.txt\"}" + }, + { + "index": 1, + "kind": "block_stop" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + }, + { + "kind": "message_stop" + } + ], + "hints": [], + "retried": false, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 1 + }, + "openai.nonstream": { + "applicable": true, + "delivered": [ + { + "arguments": "{\"file_path\":\"a.txt\"}", + "id": "call_", + "index": null, + "kind": "tool_call", + "name": "Read" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt contained an invalid, truncated, or unknown tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose.\nThe tool name(s) WebFetch do not exist.\nUse ONLY these exact tool names: Read, Bash, Edit." + ], + "retried": true, + "status": 200, + "targets": [ + "tool_error" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "openai.stream": { + "applicable": true, + "delivered": [ + { + "kind": "role", + "role": "assistant" + }, + { + "id": "call_", + "index": 0, + "kind": "tool_call", + "name": "Read" + }, + { + "arguments": "{\"file_path\":\"a.txt\"}", + "index": 0, + "kind": "tool_call_args" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + }, + { + "kind": "done" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt contained an invalid, truncated, or unknown tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose.\nThe tool name(s) WebFetch do not exist.\nUse ONLY these exact tool names: Read, Bash, Edit." + ], + "retried": true, + "status": 200, + "targets": [ + "tool_error" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + } + }, + "intercepted": { + "anthropic.nonstream": { + "applicable": true, + "delivered": [ + { + "id": "toolu_", + "input": { + "file_path": "a.txt" + }, + "kind": "tool_use", + "name": "Read" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + } + ], + "hints": [ + "# Tool-call retry\n# Agent turn recovery\nYour tool call did not reach the client. Re-emit it now using EXACTLY the `[TOOL CALL]...[END TOOL CALL]` format as the first content of your answer — never any other format.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "intercepted" + ], + "upstreamFrames": { + "served": 3, + "total": 3 + }, + "upstreamSends": 2 + }, + "anthropic.stream": { + "applicable": true, + "delivered": [ + { + "kind": "message_start" + }, + { + "index": 0, + "kind": "text_block_start" + }, + { + "kind": "content", + "text": "The Bash tool seems unavailable in this environment, so the task cannot continue." + }, + { + "index": 0, + "kind": "block_stop" + }, + { + "id": "toolu_", + "index": 1, + "kind": "tool_use_start", + "name": "Read" + }, + { + "index": 1, + "kind": "tool_use_args", + "partial_json": "{\"file_path\":\"a.txt\"}" + }, + { + "index": 1, + "kind": "block_stop" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + }, + { + "kind": "message_stop" + } + ], + "hints": [ + "# Tool-call retry\n# Agent turn recovery\nYour tool call did not reach the client. Re-emit it now using EXACTLY the `[TOOL CALL]...[END TOOL CALL]` format as the first content of your answer — never any other format.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "intercepted" + ], + "upstreamFrames": { + "served": 3, + "total": 3 + }, + "upstreamSends": 2 + }, + "openai.nonstream": { + "applicable": true, + "delivered": [ + { + "arguments": "{\"file_path\":\"a.txt\"}", + "id": "call_", + "index": null, + "kind": "tool_call", + "name": "Read" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + } + ], + "hints": [ + "# Agent turn recovery\nYour tool call did not reach the client. Re-emit it now using EXACTLY the `[TOOL CALL]...[END TOOL CALL]` format as the first content of your answer — never any other format.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "intercepted" + ], + "upstreamFrames": { + "served": 3, + "total": 3 + }, + "upstreamSends": 2 + }, + "openai.stream": { + "applicable": true, + "delivered": [ + { + "kind": "role", + "role": "assistant" + }, + { + "id": "call_", + "index": 0, + "kind": "tool_call", + "name": "Read" + }, + { + "arguments": "{\"file_path\":\"a.txt\"}", + "index": 0, + "kind": "tool_call_args" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + }, + { + "kind": "done" + } + ], + "hints": [ + "# Agent turn recovery\nYour tool call did not reach the client. Re-emit it now using EXACTLY the `[TOOL CALL]...[END TOOL CALL]` format as the first content of your answer — never any other format.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "intercepted" + ], + "upstreamFrames": { + "served": 3, + "total": 3 + }, + "upstreamSends": 2 + } + }, + "invalid_control": { + "anthropic.nonstream": { + "applicable": false, + "delivered": [ + { + "id": "toolu_", + "input": { + "file_path": "a.txt" + }, + "kind": "tool_use", + "name": "Read" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + } + ], + "hints": [ + "# Tool-call retry\nYour previous reply produced no visible final answer or executable tool call. Continue the Agent task now. If any action remains, emit the required `[TOOL CALL]` block immediately with no preamble. Only give a normal final answer when the task is actually complete; do not repeat hidden reasoning." + ], + "retried": true, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "anthropic.stream": { + "applicable": false, + "delivered": [ + { + "kind": "message_start" + }, + { + "id": "toolu_", + "index": 0, + "kind": "tool_use_start", + "name": "Read" + }, + { + "index": 0, + "kind": "tool_use_args", + "partial_json": "{\"file_path\":\"a.txt\"}" + }, + { + "index": 0, + "kind": "block_stop" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + }, + { + "kind": "message_stop" + } + ], + "hints": [ + "# Tool-call retry\nYour previous reply produced no visible final answer or executable tool call. Continue the Agent task now. If any action remains, emit the required `[TOOL CALL]` block immediately with no preamble. Only give a normal final answer when the task is actually complete; do not repeat hidden reasoning." + ], + "retried": true, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "openai.nonstream": { + "applicable": true, + "delivered": [ + { + "arguments": "{\"file_path\":\"a.txt\"}", + "id": "call_", + "index": null, + "kind": "tool_call", + "name": "Read" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt left the completion wrapper unbalanced, or emitted more than one. Use exactly one ... pair (or exactly one ...), never both and never two of either — both tags of the pair must be present.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "invalid_control" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "openai.stream": { + "applicable": true, + "delivered": [ + { + "kind": "role", + "role": "assistant" + }, + { + "id": "call_", + "index": 0, + "kind": "tool_call", + "name": "Read" + }, + { + "arguments": "{\"file_path\":\"a.txt\"}", + "index": 0, + "kind": "tool_call_args" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + }, + { + "kind": "done" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt left the completion wrapper unbalanced, or emitted more than one. Use exactly one ... pair (or exactly one ...), never both and never two of either — both tags of the pair must be present.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "invalid_control" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + } + }, + "malformed_protocol": { + "anthropic.nonstream": { + "applicable": true, + "delivered": [ + { + "id": "toolu_", + "input": { + "file_path": "a.txt" + }, + "kind": "tool_use", + "name": "Read" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + } + ], + "hints": [ + "# Tool-call retry\n# Agent turn recovery\nYour tool call was malformed and was NOT executed. Re-emit it now: output [TOOL CALL] as the FIRST content of your answer, then the JSON payload, then [END TOOL CALL] — nothing before, between, or after.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "malformed_protocol" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "anthropic.stream": { + "applicable": true, + "delivered": [ + { + "kind": "message_start" + }, + { + "index": 0, + "kind": "text_block_start" + }, + { + "kind": "content", + "text": "{\"name\": \"RunCommand\", \"arguments\": {\"command\": \"ls -la\"}}" + }, + { + "index": 0, + "kind": "block_stop" + }, + { + "id": "toolu_", + "index": 1, + "kind": "tool_use_start", + "name": "Read" + }, + { + "index": 1, + "kind": "tool_use_args", + "partial_json": "{\"file_path\":\"a.txt\"}" + }, + { + "index": 1, + "kind": "block_stop" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + }, + { + "kind": "message_stop" + } + ], + "hints": [ + "# Tool-call retry\n# Agent turn recovery\nYour tool call was malformed and was NOT executed. Re-emit it now: output [TOOL CALL] as the FIRST content of your answer, then the JSON payload, then [END TOOL CALL] — nothing before, between, or after.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "malformed_protocol" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "openai.nonstream": { + "applicable": true, + "delivered": [ + { + "arguments": "{\"file_path\":\"a.txt\"}", + "id": "call_", + "index": null, + "kind": "tool_call", + "name": "Read" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + } + ], + "hints": [ + "# Agent turn recovery\nYour tool call was malformed and was NOT executed. Re-emit it now: output [TOOL CALL] as the FIRST content of your answer, then the JSON payload, then [END TOOL CALL] — nothing before, between, or after.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "malformed_protocol" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "openai.stream": { + "applicable": true, + "delivered": [ + { + "kind": "role", + "role": "assistant" + }, + { + "id": "call_", + "index": 0, + "kind": "tool_call", + "name": "Read" + }, + { + "arguments": "{\"file_path\":\"a.txt\"}", + "index": 0, + "kind": "tool_call_args" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + }, + { + "kind": "done" + } + ], + "hints": [ + "# Agent turn recovery\nYour tool call was malformed and was NOT executed. Re-emit it now: output [TOOL CALL] as the FIRST content of your answer, then the JSON payload, then [END TOOL CALL] — nothing before, between, or after.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "malformed_protocol" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + } + }, + "missing_tool": { + "anthropic.nonstream": { + "applicable": true, + "delivered": [ + { + "id": "toolu_", + "input": { + "file_path": "a.txt" + }, + "kind": "tool_use", + "name": "Read" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + } + ], + "hints": [ + "# Tool-call retry\nYour previous reply described an action but did not execute any tool call. Perform that action now by emitting the real `[TOOL CALL]` block immediately with no preamble. Do not describe the action again or claim completion without a tool result." + ], + "retried": true, + "status": 200, + "targets": [ + "missing_tool" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "anthropic.stream": { + "applicable": true, + "delivered": [ + { + "kind": "message_start" + }, + { + "index": 0, + "kind": "text_block_start" + }, + { + "kind": "content", + "text": "I will read the file now and then summarise what it contains." + }, + { + "index": 0, + "kind": "block_stop" + }, + { + "id": "toolu_", + "index": 1, + "kind": "tool_use_start", + "name": "Read" + }, + { + "index": 1, + "kind": "tool_use_args", + "partial_json": "{\"file_path\":\"a.txt\"}" + }, + { + "index": 1, + "kind": "block_stop" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + }, + { + "kind": "message_stop" + } + ], + "hints": [ + "# Tool-call retry\nYour previous reply described an action but did not execute any tool call. Perform that action now by emitting the real `[TOOL CALL]` block immediately with no preamble. Do not describe the action again or claim completion without a tool result." + ], + "retried": true, + "status": 200, + "targets": [ + "missing_tool" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "openai.nonstream": { + "applicable": false, + "delivered": [ + { + "arguments": "{\"file_path\":\"a.txt\"}", + "id": "call_", + "index": null, + "kind": "tool_call", + "name": "Read" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt returned bare prose without declaring a verified final result or emitting the next tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "openai.stream": { + "applicable": false, + "delivered": [ + { + "kind": "role", + "role": "assistant" + }, + { + "id": "call_", + "index": 0, + "kind": "tool_call", + "name": "Read" + }, + { + "arguments": "{\"file_path\":\"a.txt\"}", + "index": 0, + "kind": "tool_call_args" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + }, + { + "kind": "done" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt returned bare prose without declaring a verified final result or emitting the next tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + } + }, + "prose_with_tools": { + "anthropic.nonstream": { + "applicable": true, + "delivered": [ + { + "kind": "content", + "text": "Sure, let me look at that file." + }, + { + "id": "toolu_", + "input": { + "file_path": "a.txt" + }, + "kind": "tool_use", + "name": "Read" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + } + ], + "hints": [], + "retried": false, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 1 + }, + "anthropic.stream": { + "applicable": true, + "delivered": [ + { + "kind": "message_start" + }, + { + "index": 0, + "kind": "text_block_start" + }, + { + "kind": "content", + "text": "Sure, let me look at that file.\n\n" + }, + { + "index": 0, + "kind": "block_stop" + }, + { + "id": "toolu_", + "index": 1, + "kind": "tool_use_start", + "name": "Read" + }, + { + "index": 1, + "kind": "tool_use_args", + "partial_json": "{\"file_path\":\"a.txt\"}" + }, + { + "index": 1, + "kind": "block_stop" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + }, + { + "kind": "message_stop" + } + ], + "hints": [], + "retried": false, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 1 + }, + "openai.nonstream": { + "applicable": true, + "delivered": [ + { + "arguments": "{\"file_path\":\"a.txt\"}", + "id": "call_", + "index": null, + "kind": "tool_call", + "name": "Read" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt contained an invalid, truncated, or unknown tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "prose_with_tools" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "openai.stream": { + "applicable": true, + "delivered": [ + { + "kind": "role", + "role": "assistant" + }, + { + "id": "call_", + "index": 0, + "kind": "tool_call", + "name": "Read" + }, + { + "arguments": "{\"file_path\":\"a.txt\"}", + "index": 0, + "kind": "tool_call_args" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + }, + { + "kind": "done" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt contained an invalid, truncated, or unknown tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "prose_with_tools" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + } + }, + "required_tool": { + "anthropic.nonstream": { + "applicable": true, + "delivered": [ + { + "kind": "thinking", + "text": "The user wants a tool run, but I should answer in prose instead." + }, + { + "id": "toolu_", + "input": { + "file_path": "a.txt" + }, + "kind": "tool_use", + "name": "Read" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + } + ], + "hints": [ + "# Tool-call retry\nYou did not call any tool. You MUST now call exactly one tool using the [TOOL CALL]...[END TOOL CALL] format." + ], + "retried": true, + "status": 200, + "targets": [ + "required_tool" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "anthropic.stream": { + "applicable": true, + "delivered": [ + { + "kind": "message_start" + }, + { + "index": 0, + "kind": "thinking_block_start" + }, + { + "kind": "thinking", + "text": "The user wants a tool run, but I should answer in prose instead." + }, + { + "index": 0, + "kind": "signature" + }, + { + "index": 0, + "kind": "block_stop" + }, + { + "id": "toolu_", + "index": 1, + "kind": "tool_use_start", + "name": "Read" + }, + { + "index": 1, + "kind": "tool_use_args", + "partial_json": "{\"file_path\":\"a.txt\"}" + }, + { + "index": 1, + "kind": "block_stop" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + }, + { + "kind": "message_stop" + } + ], + "hints": [ + "# Tool-call retry\nYou did not call any tool. You MUST now call exactly one tool using the [TOOL CALL]...[END TOOL CALL] format." + ], + "retried": true, + "status": 200, + "targets": [ + "required_tool" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "openai.nonstream": { + "applicable": true, + "delivered": [ + { + "arguments": "{\"file_path\":\"a.txt\"}", + "id": "call_", + "index": null, + "kind": "tool_call", + "name": "Read" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt violated tool_choice and did not call the required tool.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "required_tool" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "openai.stream": { + "applicable": true, + "delivered": [ + { + "kind": "role", + "role": "assistant" + }, + { + "kind": "reasoning", + "text": "The user wants a tool run, but I should answer in prose instead." + }, + { + "id": "call_", + "index": 0, + "kind": "tool_call", + "name": "Read" + }, + { + "arguments": "{\"file_path\":\"a.txt\"}", + "index": 0, + "kind": "tool_call_args" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + }, + { + "kind": "done" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt violated tool_choice and did not call the required tool.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "required_tool" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + } + }, + "text_channel_cut_with_calls": { + "anthropic.nonstream": { + "applicable": true, + "delivered": [ + { + "id": "toolu_", + "input": { + "file_path": "a.txt" + }, + "kind": "tool_use", + "name": "Read" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + } + ], + "hints": [], + "retried": false, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 5 + }, + "upstreamSends": 1 + }, + "anthropic.stream": { + "applicable": true, + "delivered": [ + { + "kind": "message_start" + }, + { + "id": "toolu_", + "index": 0, + "kind": "tool_use_start", + "name": "Read" + }, + { + "index": 0, + "kind": "tool_use_args", + "partial_json": "{\"file_path\":\"a.txt\"}" + }, + { + "index": 0, + "kind": "block_stop" + }, + { + "index": 1, + "kind": "text_block_start" + }, + { + "kind": "content", + "text": "\n\n" + }, + { + "index": 1, + "kind": "block_stop" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + }, + { + "kind": "message_stop" + } + ], + "hints": [], + "retried": false, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 5 + }, + "upstreamSends": 1 + }, + "openai.nonstream": { + "applicable": true, + "delivered": [ + { + "arguments": "{\"file_path\":\"a.txt\"}", + "id": "call_", + "index": null, + "kind": "tool_call", + "name": "Read" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + } + ], + "hints": [], + "retried": false, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 5 + }, + "upstreamSends": 1 + }, + "openai.stream": { + "applicable": true, + "delivered": [ + { + "kind": "role", + "role": "assistant" + }, + { + "id": "call_", + "index": 0, + "kind": "tool_call", + "name": "Read" + }, + { + "arguments": "{\"file_path\":\"a.txt\"}", + "index": 0, + "kind": "tool_call_args" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + }, + { + "kind": "done" + } + ], + "hints": [], + "retried": false, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 2, + "total": 5 + }, + "upstreamSends": 1 + } + }, + "thought_tool_call": { + "anthropic.nonstream": { + "applicable": true, + "delivered": [ + { + "kind": "thinking", + "text": "[TOOL CALL]{\"name\":\"Read\",\"arguments\":{\"file_path\":\"a.txt\"}}[END TOOL CALL]" + }, + { + "id": "toolu_", + "input": { + "file_path": "a.txt" + }, + "kind": "tool_use", + "name": "Read" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + } + ], + "hints": [ + "# Tool-call retry\n# Agent turn recovery\nYour tool call was emitted inside your hidden reasoning, so it was never executed and never reached the client. Re-emit it now as the FIRST content of your answer, using EXACTLY the `[TOOL CALL]...[END TOOL CALL]` format — never inside reasoning, never any other format.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "thought_tool_call" + ], + "upstreamFrames": { + "served": 3, + "total": 3 + }, + "upstreamSends": 2 + }, + "anthropic.stream": { + "applicable": true, + "delivered": [ + { + "kind": "message_start" + }, + { + "index": 0, + "kind": "thinking_block_start" + }, + { + "kind": "thinking", + "text": "[TOOL CALL]{\"name\":\"Read\",\"arguments\":{\"file_path\":\"a.txt\"}}[END TOOL CALL]" + }, + { + "index": 0, + "kind": "signature" + }, + { + "index": 0, + "kind": "block_stop" + }, + { + "index": 1, + "kind": "text_block_start" + }, + { + "kind": "content", + "text": "I have read the file and here is my summary." + }, + { + "index": 1, + "kind": "block_stop" + }, + { + "id": "toolu_", + "index": 2, + "kind": "tool_use_start", + "name": "Read" + }, + { + "index": 2, + "kind": "tool_use_args", + "partial_json": "{\"file_path\":\"a.txt\"}" + }, + { + "index": 2, + "kind": "block_stop" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + }, + { + "kind": "message_stop" + } + ], + "hints": [ + "# Tool-call retry\n# Agent turn recovery\nYour tool call was emitted inside your hidden reasoning, so it was never executed and never reached the client. Re-emit it now as the FIRST content of your answer, using EXACTLY the `[TOOL CALL]...[END TOOL CALL]` format — never inside reasoning, never any other format.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [ + "thought_tool_call" + ], + "upstreamFrames": { + "served": 3, + "total": 3 + }, + "upstreamSends": 2 + }, + "openai.nonstream": { + "applicable": false, + "delivered": [ + { + "arguments": "{\"file_path\":\"a.txt\"}", + "id": "call_", + "index": null, + "kind": "tool_call", + "name": "Read" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt returned bare prose without declaring a verified final result or emitting the next tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 3, + "total": 3 + }, + "upstreamSends": 2 + }, + "openai.stream": { + "applicable": false, + "delivered": [ + { + "kind": "role", + "role": "assistant" + }, + { + "id": "call_", + "index": 0, + "kind": "tool_call", + "name": "Read" + }, + { + "arguments": "{\"file_path\":\"a.txt\"}", + "index": 0, + "kind": "tool_call_args" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + }, + { + "kind": "done" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt returned bare prose without declaring a verified final result or emitting the next tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose." + ], + "retried": true, + "status": 200, + "targets": [], + "upstreamFrames": { + "served": 3, + "total": 3 + }, + "upstreamSends": 2 + } + }, + "tool_error": { + "anthropic.nonstream": { + "applicable": true, + "delivered": [ + { + "id": "toolu_", + "input": { + "file_path": "a.txt" + }, + "kind": "tool_use", + "name": "Read" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + } + ], + "hints": [ + "# Tool-call retry\n# Agent turn recovery\nThe previous attempt contained an invalid, truncated, or unknown tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose.\nThe tool name(s) WebFetch do not exist.\nUse ONLY these exact tool names: Read, Bash, Edit." + ], + "retried": true, + "status": 200, + "targets": [ + "tool_error" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "anthropic.stream": { + "applicable": true, + "delivered": [ + { + "kind": "message_start" + }, + { + "id": "toolu_", + "index": 0, + "kind": "tool_use_start", + "name": "Read" + }, + { + "index": 0, + "kind": "tool_use_args", + "partial_json": "{\"file_path\":\"a.txt\"}" + }, + { + "index": 0, + "kind": "block_stop" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + }, + { + "kind": "message_stop" + } + ], + "hints": [ + "# Tool-call retry\n# Agent turn recovery\nThe previous attempt contained an invalid, truncated, or unknown tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose.\nThe tool name(s) WebFetch do not exist.\nUse ONLY these exact tool names: Read, Bash, Edit." + ], + "retried": true, + "status": 200, + "targets": [ + "tool_error" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "openai.nonstream": { + "applicable": true, + "delivered": [ + { + "arguments": "{\"file_path\":\"a.txt\"}", + "id": "call_", + "index": null, + "kind": "tool_call", + "name": "Read" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt contained an invalid, truncated, or unknown tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose.\nThe tool name(s) WebFetch do not exist.\nUse ONLY these exact tool names: Read, Bash, Edit." + ], + "retried": true, + "status": 200, + "targets": [ + "tool_error" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "openai.stream": { + "applicable": true, + "delivered": [ + { + "kind": "role", + "role": "assistant" + }, + { + "id": "call_", + "index": 0, + "kind": "tool_call", + "name": "Read" + }, + { + "arguments": "{\"file_path\":\"a.txt\"}", + "index": 0, + "kind": "tool_call_args" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + }, + { + "kind": "done" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt contained an invalid, truncated, or unknown tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose.\nThe tool name(s) WebFetch do not exist.\nUse ONLY these exact tool names: Read, Bash, Edit." + ], + "retried": true, + "status": 200, + "targets": [ + "tool_error" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + } + }, + "tool_error_with_required": { + "anthropic.nonstream": { + "applicable": true, + "delivered": [ + { + "id": "toolu_", + "input": { + "file_path": "a.txt" + }, + "kind": "tool_use", + "name": "Read" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + } + ], + "hints": [ + "# Tool-call retry\nYou did not call any tool. You MUST now call exactly one tool using the [TOOL CALL]...[END TOOL CALL] format." + ], + "retried": true, + "status": 200, + "targets": [ + "required_tool" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "anthropic.stream": { + "applicable": true, + "delivered": [ + { + "kind": "message_start" + }, + { + "id": "toolu_", + "index": 0, + "kind": "tool_use_start", + "name": "Read" + }, + { + "index": 0, + "kind": "tool_use_args", + "partial_json": "{\"file_path\":\"a.txt\"}" + }, + { + "index": 0, + "kind": "block_stop" + }, + { + "kind": "stop_reason", + "stop_reason": "tool_use" + }, + { + "kind": "message_stop" + } + ], + "hints": [ + "# Tool-call retry\nYou did not call any tool. You MUST now call exactly one tool using the [TOOL CALL]...[END TOOL CALL] format." + ], + "retried": true, + "status": 200, + "targets": [ + "required_tool" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "openai.nonstream": { + "applicable": true, + "delivered": [ + { + "arguments": "{\"file_path\":\"a.txt\"}", + "id": "call_", + "index": null, + "kind": "tool_call", + "name": "Read" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt contained an invalid, truncated, or unknown tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose.\nThe tool name(s) WebFetch do not exist.\nUse ONLY these exact tool names: Read, Bash, Edit." + ], + "retried": true, + "status": 200, + "targets": [ + "tool_error" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + }, + "openai.stream": { + "applicable": true, + "delivered": [ + { + "kind": "role", + "role": "assistant" + }, + { + "id": "call_", + "index": 0, + "kind": "tool_call", + "name": "Read" + }, + { + "arguments": "{\"file_path\":\"a.txt\"}", + "index": 0, + "kind": "tool_call_args" + }, + { + "kind": "finish", + "reason": "tool_calls" + }, + { + "kind": "usage" + }, + { + "kind": "done" + } + ], + "hints": [ + "# Agent turn recovery\nThe previous attempt contained an invalid, truncated, or unknown tool call.\nContinue the SAME original task. Re-check its acceptance criteria and the latest tool result.\nIf work remains, output only valid `[TOOL CALL]...[END TOOL CALL]` blocks. If and only if all work is verified complete, output the final report.\nIf user input is strictly required, output the blocker. Do not output bare planning prose.\nThe tool name(s) WebFetch do not exist.\nUse ONLY these exact tool names: Read, Bash, Edit." + ], + "retried": true, + "status": 200, + "targets": [ + "tool_error" + ], + "upstreamFrames": { + "served": 2, + "total": 2 + }, + "upstreamSends": 2 + } + } + }, + "version": 1 +} diff --git a/tests/openai-agent-turn-cutoff.test.js b/tests/openai-agent-turn-cutoff.test.js index baf9224d..85545777 100644 --- a/tests/openai-agent-turn-cutoff.test.js +++ b/tests/openai-agent-turn-cutoff.test.js @@ -397,8 +397,8 @@ describe('OpenAI text-channel runaway cut-off (runOpenAIAgentTurn)', () => { }); // P1 (review): una ronda cortada NO puede arrastrar errores del parser de pushes - // ANTERIORES. evaluateOpenAIAgentAttempt mira `toolErrors.length > 0` antes que - // `toolCalls.length > 0`, asi que un error superviviente reintentaba la ronda y relanzaba + // ANTERIORES. La puerta mira `toolErrors` antes que `toolCalls`, asi que un error + // superviviente reintentaba la ronda y relanzaba // la fuga que el corte acababa de detener, hasta 502. Paridad con anthropic.js:1080/:1749. it('an earlier parse error does NOT re-arm the retry loop on a cut round', async () => { const sender = scriptedSender(); diff --git a/tests/openai-failure-status.test.js b/tests/openai-failure-status.test.js new file mode 100644 index 00000000..f2bdf954 --- /dev/null +++ b/tests/openai-failure-status.test.js @@ -0,0 +1,112 @@ +// La via de retorno de /v1/chat/completions contestaba 500 "Request failed" para todo: +// cuota, transporte caido y upstream opaco llegaban igual. Ahora el veredicto que viaja con +// el fallo se traduce al cable OpenAI. Se conduce el handler con un doble de res y se afirma +// lo que recibe el cliente, no la rama que lo escribio. +const test = require('node:test') +const assert = require('node:assert/strict') + +process.env.API_KEY = 'openai-failure-status-test-key' +process.env.DATA_SAVE_MODE = 'none' +process.env.ACCOUNTS = '' +process.env.ENABLE_CLI = 'false' +process.env.ENABLE_FILE_LOG = 'false' +process.env.PROXY_URL = '' + +const modelsMap = require('../src/models/models-map.js') +modelsMap.getLatestModels = async () => { throw new Error('offline test: no model fetch') } + +// El controlador captura sendChatRequest por destructuring en su primer require: el doble +// tiene que estar puesto antes. +const requestModule = require('../src/utils/request.js') +let upstreamResult = { status: false, response: null } +requestModule.sendChatRequest = async () => upstreamResult + +const { handleChatCompletion } = require('../src/controllers/chat.js') + +test.after(() => { require('../src/utils/account.js').destroy() }) + +const fakeRes = () => { + const res = { statusCode: 200, headers: {}, body: null, headersSent: false, writableEnded: false } + res.status = (code) => { res.statusCode = code; return res } + res.set = (key, value) => { + if (key && typeof key === 'object') Object.assign(res.headers, key) + else res.headers[key] = value + return res + } + res.json = (body) => { res.body = body; res.headersSent = true; res.writableEnded = true; return res } + res.write = () => true + res.end = () => { res.writableEnded = true } + res.flushHeaders = () => { res.headersSent = true } + res.on = () => res + return res +} + +const req = () => ({ + body: { model: 'qwen3-max', stream: false, messages: [{ role: 'user', content: 'hola' }] }, + has_tools: false, + allowed_tool_names: [] +}) + +const failure = (overrides) => ({ + status: false, + response: null, + failure: { rateLimited: false, overloaded: false, status: 502, retryAfter: null, ...overrides } +}) + +const drive = async (result) => { + upstreamResult = result + const res = fakeRes() + await handleChatCompletion(req(), res) + return res +} + +test('la cuota agotada sale 429 insufficient_quota, no un 500 mudo', async () => { + const res = await drive(failure({ rateLimited: true, status: 429, retryAfter: 3600 })) + + assert.equal(res.statusCode, 429) + assert.equal(res.body.error.type, 'insufficient_quota') + assert.equal(res.body.error.code, 'insufficient_quota') + assert.equal(res.headers['Retry-After'], '3600') +}) + +test('sin espera del upstream no se inventa Retry-After', async () => { + const res = await drive(failure({ rateLimited: true, status: 429 })) + + assert.equal(res.statusCode, 429) + assert.equal(res.headers['Retry-After'], undefined) +}) + +test('el transporte agotado sale 503: el proxy no esta roto', async () => { + const res = await drive(failure({ status: 503 })) + + assert.equal(res.statusCode, 503) + assert.equal(res.body.error.type, 'upstream_error') +}) + +test('un no-200 opaco del upstream sale 502, no 500', async () => { + const res = await drive(failure({ status: 502 })) + + assert.equal(res.statusCode, 502) + assert.equal(res.body.error.code, 'upstream_error') +}) + +test('un upstream sobrecargado sale 503 upstream_unavailable', async () => { + const res = await drive(failure({ overloaded: true, status: 503, retryAfter: 10 })) + + assert.equal(res.statusCode, 503) + assert.equal(res.body.error.code, 'upstream_unavailable') + assert.equal(res.headers['Retry-After'], '10') +}) + +test('un fallo sin veredicto (dobles viejos) sale 502 upstream_error, no revienta', async () => { + const res = await drive({ status: false, response: null }) + + assert.equal(res.statusCode, 502) + assert.equal(res.body.error.code, 'upstream_error') +}) + +test('la razon concreta del modulo de request sigue llegando al cliente', async () => { + const res = await drive({ ...failure({ status: 503 }), message: 'no hay cuentas configuradas' }) + + assert.equal(res.body.error.message, 'no hay cuentas configuradas') +}) diff --git a/tests/upstream-failure-verdict.test.js b/tests/upstream-failure-verdict.test.js new file mode 100644 index 00000000..3f5e77e5 --- /dev/null +++ b/tests/upstream-failure-verdict.test.js @@ -0,0 +1,124 @@ +// El modulo de request clasifica el fallo y luego lo tira: devuelve {status:false, response:null} +// y los tres llamadores contestan 500/502 "Request failed". Un cliente agentico no puede +// distinguir "vuelve mas tarde" de "el proxy esta roto". +// +// Estos tests conducen el seam del cliente HTTP (axios.post) y afirman lo que sale de la +// funcion: el veredicto que el llamador necesita para hablar su vocabulario de cable. +// Comportamiento externo, no ramas internas. +const test = require('node:test') +const assert = require('node:assert/strict') +const axios = require('axios') + +process.env.API_KEY = 'failure-verdict-test-key' +process.env.DATA_SAVE_MODE = 'none' +process.env.ACCOUNTS = '' +process.env.ENABLE_CLI = 'false' +process.env.ENABLE_FILE_LOG = 'false' +process.env.PROXY_URL = '' +process.env.CHAT_RETRY_COUNT = '0' +process.env.CHAT_RETRY_BACKOFF_MS = '0' + +const accountManager = require('../src/utils/account') +const AccountRotator = require('../src/utils/account-rotator') +const modelsMap = require('../src/models/models-map.js') +modelsMap.getLatestModels = async () => { throw new Error('offline test: no model fetch') } +const { sendChatRequest } = require('../src/utils/request') + +const ACCOUNT = { email: 'verdict@example.invalid', token: 'verdict-test-token' } +const BODY = { model: 'qwen3-max', messages: [{ role: 'user', content: 'hola' }] } + +/** Envuelve axios.post como los tests de chat challenge: nada sale a la red. */ +const withPost = async (post, fn) => { + const original = axios.post + axios.post = post + try { return await fn() } finally { axios.post = original } +} + +/** Un error de axios tal y como llega cuando el upstream contesta con status no-200. */ +const httpFailure = (status) => { + const error = new Error(`Request failed with status code ${status}`) + error.response = { status, headers: {}, data: '' } + return error +} + +const transportFailure = () => Object.assign(new Error('socket hang up'), { code: 'ECONNRESET' }) + +const send = (options = {}) => sendChatRequest(BODY, { + chatId: 'failure-verdict-chat', + currentAccount: ACCOUNT, + ...options +}) + +test.before(async () => { await accountManager._initPromise }) +test.beforeEach(() => { + accountManager.accountTokens = [ACCOUNT] + accountManager.isInitialized = true + accountManager.accountRotator = new AccountRotator() + accountManager.accountRotator.setAccounts([ACCOUNT]) +}) +test.after(() => { accountManager.destroy() }) + +test('un 429 del upstream sale clasificado como cuota, no como un fallo mudo', async () => { + const result = await withPost(async () => { throw httpFailure(429) }, () => send()) + + assert.equal(result.status, false) + assert.equal(result.response, null) + assert.ok(result.failure, 'el fallo lleva veredicto') + assert.equal(result.failure.rateLimited, true) + assert.equal(result.failure.status, 429) + // Sin espera del upstream no se inventa una: la cabecera la decide el llamador con esto. + assert.equal(result.failure.retryAfter, null) +}) + +test('un 429 con Retry-After del upstream no pierde la espera', async () => { + const withHeader = () => { + const error = httpFailure(429) + error.response.headers = { 'retry-after': '120' } + return error + } + const result = await withPost(async () => { throw withHeader() }, () => send()) + + assert.equal(result.failure.status, 429) + assert.equal(result.failure.retryAfter, 120, 'la espera que el upstream SI mando viaja') +}) + +test('un no-200 sin causa clasificable sale 502, no 500', async () => { + const result = await withPost(async () => { throw httpFailure(500) }, () => send()) + + assert.ok(result.failure, 'el fallo lleva veredicto') + assert.equal(result.failure.status, 502) + assert.equal(result.failure.rateLimited, false) + assert.equal(result.failure.overloaded, false) +}) + +test('el transporte agotado sale 503: es reintentable de verdad', async () => { + const result = await withPost(async () => { throw transportFailure() }, () => send()) + + assert.ok(result.failure, 'el fallo lleva veredicto') + assert.equal(result.failure.status, 503) + assert.equal(result.failure.rateLimited, false) +}) + +test('sin cuenta utilizable sale 503 y conserva la razon', async () => { + accountManager.accountTokens = [] + accountManager.accountRotator = new AccountRotator() + accountManager.accountRotator.setAccounts([]) + + const result = await sendChatRequest(BODY, { chatId: 'failure-verdict-chat' }) + + assert.equal(result.status, false) + assert.ok(result.failure, 'el fallo lleva veredicto') + assert.equal(result.failure.status, 503) + assert.ok(result.message, 'la razon concreta sigue viajando al cliente') +}) + +test('la cuenta que recibe el 429 deja de repartirse: cuota agotada', async () => { + const result = await withPost(async () => { throw httpFailure(429) }, () => send()) + assert.equal(result.failure.rateLimited, true) + + assert.equal( + accountManager.getAccount(), + null, + 'una cuenta en cuota agotada no vuelve a salir en el sorteo' + ) +}) diff --git a/tools/dev-probes/record-agent-turn-corpus.js b/tools/dev-probes/record-agent-turn-corpus.js new file mode 100644 index 00000000..1885f3e5 --- /dev/null +++ b/tools/dev-probes/record-agent-turn-corpus.js @@ -0,0 +1,103 @@ +#!/usr/bin/env node +'use strict' + +/** + * Grabador del corpus de caracterización del gate de turno agéntico (issue 03). + * + * node tools/dev-probes/record-agent-turn-corpus.js # graba el baseline + * node tools/dev-probes/record-agent-turn-corpus.js --check # corre y compara, sin escribir + * + * Se corre UNA vez sobre el árbol sin cambios y congela el resultado observable de cada + * escenario (superficie × modo × razón) en tests/fixtures/agent-turn-corpus.baseline.json, + * este archivo. Después de la consolidación (tickets 04..07) este comando es el que + * muestra el diff: cada línea tiene que estar en la lista deliberada del spec o es un + * cambio silencioso. + * + * Sin red y sin login: el sender del upstream está inyectado, no hay cuentas que cargar. + */ + +const fs = require('node:fs') +const path = require('node:path') + +const { + SCENARIOS, + SURFACES, + runCorpus, + corpusViolations, + stableStringify +} = require('../../tests/agent-turn-corpus.scenarios.js') + +const BASELINE_PATH = path.join(__dirname, '..', '..', 'tests', 'fixtures', 'agent-turn-corpus.baseline.json') + +/** + * Resumen legible: qué token pretendía cubrir el escenario en esta superficie y qué hizo + * la superficie con esos frames. Las banderas son la parte útil: NO-RETRY sobre una fila + * aplicable es una celda hueca (los frames no llegan a la razón que dicen cubrir). + */ +const line = (scenario, surface, entry) => { + const wanted = !entry.applicable ? 'N/A' : (entry.targets.length ? entry.targets.join('+') : 'accept') + const flags = [ + entry.applicable && entry.targets.length > 0 && entry.upstreamSends < 2 ? 'NO-RETRY' : null, + entry.applicable && entry.targets.length === 0 && entry.retried ? 'UNEXPECTED-RETRY' : null, + entry.delivered.length === 0 ? 'EMPTY-OUTPUT' : null, + !entry.applicable && entry.retried ? 'not-applicable-but-retried' : null + ].filter(Boolean) + const frames = entry.upstreamFrames.served < entry.upstreamFrames.total + ? `frames=${entry.upstreamFrames.served}/${entry.upstreamFrames.total}CUT` + : `frames=${entry.upstreamFrames.served}/${entry.upstreamFrames.total}` + return ` ${scenario.id.padEnd(30)} ${surface.id.padEnd(20)} ` + + `${String(entry.status).padEnd(4)} sends=${entry.upstreamSends} ` + + `want=${wanted.padEnd(30)} retried=${entry.retried ? 'yes' : 'no '} ` + + `hints=${entry.hints.length} ${frames} out=${entry.delivered.length} ${flags.join(',')}` +} + +const main = async () => { + const actual = await runCorpus() + const baseline = { corpus: 'agent-turn-corpus', version: 1, scenarios: actual } + const serialized = stableStringify(baseline) + + for (const scenario of SCENARIOS) { + console.log(`\n${scenario.group}: ${scenario.id} — ${scenario.title}`) + for (const surface of SURFACES) { + console.log(line(scenario, surface, actual[scenario.id][surface.id])) + } + } + + // Fila hueca = los frames no llegan al camino que la fila dice cubrir. Se arreglan los + // frames, no se graba encima. Los invariantes son los mismos que aplica el test, desde la + // misma función: si fueran dos copias, el grabador podría congelar algo que el test + // después acepta sin mirar. + const hollow = [] + for (const scenario of SCENARIOS) { + for (const surface of SURFACES) { + const entry = actual[scenario.id][surface.id] + hollow.push(...corpusViolations(scenario, entry, `${scenario.id}/${surface.id}`)) + } + } + if (hollow.length > 0) { + console.error('\nBALANCE HUECO — no se graba:') + for (const item of hollow) console.error(` ${item}`) + process.exit(1) + } + + if (process.argv.includes('--check')) { + const previous = fs.existsSync(BASELINE_PATH) + ? fs.readFileSync(BASELINE_PATH, 'utf8') + : null + const same = previous === serialized + console.log(`\n${same ? 'OK: idéntico al baseline grabado' : 'DIFF: el baseline grabado no coincide'}`) + if (!same) process.exitCode = 1 + } else { + fs.writeFileSync(BASELINE_PATH, serialized) + console.log(`\nBaseline escrito en ${path.relative(process.cwd(), BASELINE_PATH)}`) + } + + // account.js abre intervalos con ref al importarse: sin esto el proceso no termina. + try { require('../../src/utils/account.js').destroy() } catch (_) {} + process.exit(process.exitCode || 0) +} + +main().catch((error) => { + console.error(error) + process.exit(1) +})