From f78871d754a9e94fa5940310276a8ae0d8aed236 Mon Sep 17 00:00:00 2001
From: Jason Wang <56037774+JasonW404@users.noreply.github.com>
Date: Sun, 20 Sep 2026 20:47:33 +0800
Subject: [PATCH] [fix] enforce explicit CodeAgent termination and silent
recovery (#3969)
* fix(agent): enforce explicit CodeAgent termination
* fix(test): restore CodeAgent CI compatibility
---
backend/agents/nl2skill_agent.py | 1 +
backend/prompts/nl2agent_en.yaml | 18 +-
backend/prompts/nl2agent_zh.yaml | 20 +-
.../prompts/skill_creation_complicate_en.yaml | 9 +-
.../prompts/skill_creation_complicate_zh.yaml | 9 +-
backend/prompts/skill_creation_simple_en.yaml | 7 +-
backend/prompts/skill_creation_simple_zh.yaml | 7 +-
backend/prompts/utils/prompt_generate_en.yaml | 40 +-
backend/prompts/utils/prompt_generate_zh.yaml | 40 +-
backend/utils/content_classifier_utils.py | 12 +-
backend/utils/context_utils.py | 4 +-
doc/docs/en/user-guide/start-chat.md | 2 +-
doc/docs/zh/user-guide/start-chat.md | 2 +-
.../agentConfig/SkillBuildModal.tsx | 7 +
.../adapter/remote-chat-model-adapter.ts | 72 +++
.../tests/nl2SkillAttemptRollback.test.ts | 26 +
.../configs/gsm8k_solver_assistant.yaml | 2 +-
sdk/nexent/core/agents/agent_model.py | 4 +
sdk/nexent/core/agents/core_agent.py | 482 +++++++++++-------
sdk/nexent/core/agents/nexent_agent.py | 7 +-
sdk/nexent/core/agents/output_protocol.py | 204 ++++++++
sdk/nexent/core/models/openai_llm.py | 22 +-
.../core/tools/create_scheduled_task_tool.py | 6 +-
test/assets/test_prompt.yaml | 6 +-
test/assets/test_sub_prompt.yaml | 6 +-
.../backend/services/test_nl2skill_service.py | 7 +-
.../utils/test_content_classifier_utils.py | 26 +
test/sdk/core/agents/test_agent_model.py | 1 +
test/sdk/core/agents/test_core_agent.py | 443 ++++++++--------
.../core/agents/test_core_agent_planning.py | 19 +-
.../core/agents/test_guardrail_checkpoints.py | 122 ++++-
test/sdk/core/agents/test_nexent_agent.py | 24 +
test/sdk/core/agents/test_output_protocol.py | 113 ++++
test/sdk/core/models/test_openai_llm.py | 28 +
34 files changed, 1307 insertions(+), 491 deletions(-)
create mode 100644 frontend/tests/nl2SkillAttemptRollback.test.ts
create mode 100644 sdk/nexent/core/agents/output_protocol.py
create mode 100644 test/sdk/core/agents/test_output_protocol.py
diff --git a/backend/agents/nl2skill_agent.py b/backend/agents/nl2skill_agent.py
index cbb4d6cec..3729d0137 100644
--- a/backend/agents/nl2skill_agent.py
+++ b/backend/agents/nl2skill_agent.py
@@ -19,6 +19,7 @@ def create_nl2skill_agent_config(
tools=[],
max_steps=5,
model_name=model_name,
+ output_protocol="final_answer_envelope",
provide_run_summary=False,
instructions=system_prompt,
enable_planning=False,
diff --git a/backend/prompts/nl2agent_en.yaml b/backend/prompts/nl2agent_en.yaml
index 7dadec4be..c49b60689 100644
--- a/backend/prompts/nl2agent_en.yaml
+++ b/backend/prompts/nl2agent_en.yaml
@@ -18,7 +18,7 @@ system_prompt: |-
- Only when the current input is a `full_generation` `requirement_clarification` submission may you use its answers to continue saving the description. Partial field and resource tasks may use only information explicitly provided for the current task and facts from the verified draft.
- The Agent display name, generated variable name, verbs such as "create" or "generate", and common domain knowledge are not confirmed requirements. Never infer the task, users, output, or constraints from them.
- This completion rule applies only to initial full generation. Configuration is complete only after the description is saved, resource requirements are installed and bound or explicitly abandoned, every Prompt field is saved, and an `agent_generation_completed` state event is received.
- - `full_generation` is the only task subject to the initial full-generation completion restriction: before receiving `agent_generation_completed`, do not output a plain completion statement, simulate execution results, or say "I created it" or "already created"; every turn must advance the full-generation state through a business tool or wrapper. A partial task may output its local result or a Revision Summary when complete, but must not claim that the Agent has completed generation.
+ - `full_generation` is the only task subject to the initial full-generation completion restriction: before receiving `agent_generation_completed`, do not return a completion statement, simulate execution results, or say "I created it" or "already created"; every turn must advance the full-generation state through a business tool or wrapper. A partial task may return its local result or a Revision Summary through `final_answer(...)` when complete, but must not claim that the Agent has completed generation.
### Intent And Minimal Workflow
- First determine the task the user wants to complete in this turn, then use the current `nl2agent_verified_state` to select the smallest executable workflow. Do not label the entire conversation as "generation mode" or "revision mode" before identifying the task.
@@ -41,7 +41,7 @@ system_prompt: |-
- After a resource task receives a `suggested_resource_installation` action, continue the same resource task; after installation, search again for real installed resources. After receiving an `installed_resource_binding` `continue` action, enter the Resource-Dependent Prompt Generation stage using the newly injected `bound_resources`; do not output a summary directly. When no resource will be bound, skip the empty binding card and enter Prompt generation immediately using the authoritative `bound_resources`.
- Resource-Dependent Prompt Generation must atomically regenerate and save `duty_prompt`, `constraint_prompt`, and `few_shots_prompt` in that order. These fields must reflect real bound resources and their declared inputs; do not retain content that references removed resources, omits new capabilities, or invents invocation details.
- If an added, replaced, or reconfigured resource changes the Agent's purpose, opening capabilities, or the range of actionable user questions, also regenerate and save the affected `description`, `greeting_message`, or `example_questions`. Keep unaffected fields at their persisted values; each save may include only fields that are already determined to require an update.
- - After Resource-Dependent Prompt Generation completes, output the Revision Summary, naming the resource and Prompt fields actually updated. Only when the user explicitly requests resource removal, direct them to the Tools and Skills section of the form on the right.
+ - After Resource-Dependent Prompt Generation completes, return the Revision Summary through `final_answer(...)`, naming the resource and Prompt fields actually updated. Only when the user explicitly requests resource removal, direct them to the Tools and Skills section of the form on the right.
- Conversational removal is unsupported. For removal, tell the user to use the Tools and Skills section of the form on the right. For replacement, the new resource may be added first, but tell the user to remove the old resource in that form.
- Except for empty-name initialization, `name`, `display_name`, model settings, publication status, version state, and any other field outside the six generated fields are not editable through NL2Agent. Direct the user to the corresponding form on the right without calling a save or resource tool.
@@ -61,8 +61,8 @@ system_prompt: |-
2. During full generation, the latest successful tool result has `updated_fields` equal to `["duty_prompt"]`: generate and save only `constraint_prompt`.
3. During full generation, the latest successful tool result has `updated_fields` equal to `["constraint_prompt"]`: generate and save only `few_shots_prompt`.
4. During full generation, the latest successful tool result has `updated_fields` equal to `["few_shots_prompt"]`: generate and save only `greeting_message` and `example_questions` together.
- 5. During full generation, the latest successful tool result has `updated_fields` equal to `["greeting_message", "example_questions"]` and contains an `agent_generation_completed` state event: output only the plain text required by Completion Summary.
- - After a partial task saves fields or completes a resource operation, output its Revision Summary or local result directly when there is no pending card action. Never generate or save another field merely because the partial task updated `duty_prompt`, `constraint_prompt`, or any other field.
+ 5. During full generation, the latest successful tool result has `updated_fields` equal to `["greeting_message", "example_questions"]` and contains an `agent_generation_completed` state event: return only the Completion Summary through `final_answer(...)` in the single `...` action.
+ - After a partial task saves fields or completes a resource operation, return its Revision Summary or local result through `final_answer(...)` when there is no pending card action. Never generate or save another field merely because the partial task updated `duty_prompt`, `constraint_prompt`, or any other field.
- Never mention, generate, or save fields from a later branch. If a save fails, correct and retry the current branch once only.
### Variable Name Initialization
@@ -85,7 +85,7 @@ system_prompt: |-
6. After a `suggested_resource_installation` action, preserve its `installed` and `skipped` results unchanged. If `installed` is empty, treat that action as the user's explicit choice to continue without any suggested Tool or Skill: do not search for alternatives and do not call `{{ wrapper_name }}` with an empty resource list; generate and save only `duty_prompt` as the next atomic action. Otherwise, search `{{ installed_tool_name }}` again with the same requirements and trust only newly returned real `tool_id`/`skill_id` values. If requirements remain uncovered, place every installed and skipped candidate ref in `exclude_refs` while searching alternatives. When no alternative exists, use a clarification card requiring the user to revise, explicitly abandon, or end; never claim an unbound capability is available.
7. After installed search succeeds, choose the smallest candidate set covering strong matches and keep the total at or below {{ max_results }}. Already-bound Tools remain normal candidates: retain their search scores and include them when selected by the same coverage rules, so the binding card can show their current configuration for confirmation or revision. When the selected set is non-empty, pass unchanged candidates to `{{ recommend_tool_name }}`, decode its result, then pass that unchanged dictionary and the same `agent_id` to `{{ wrapper_name }}` with subtype `installed_resource_binding`. When the selected set is empty, skip both calls and generate and save only `duty_prompt`.
8. After an `installed_resource_binding` action with `continue` or `retry_generation`, use only the newly injected `bound_resources` database facts and follow the Atomic Action Contract strictly, executing only one Prompt branch per model response.
- 9. After the final Prompt batch succeeds and `agent_generation_completed` is received, output the plain-text completion summary directly. Do not call another tool or wrapper. For tool errors, use only `code` and `retryable`; retry at most once.
+ 9. After the final Prompt batch succeeds and `agent_generation_completed` is received, call `final_answer(completion_summary)` inside one literal `...` block. Do not call another business tool or wrapper. For tool errors, use only `code` and `retryable`; retry at most once.
### Clarification Schema
Each question contains a stable `question_id`, `question_type` (`single_choice`, `multiple_choice`, or `text`), concise `title`, and `required`. Choice questions include `options` and set `allow_other=True` and `other_input_expanded=True`. Text questions set both fields to `False` because their primary input is already open text.
@@ -200,10 +200,10 @@ system_prompt: |-
If a Prompt save fails, correct and retry that batch once. On a second failure, stop without a completion summary; successfully saved fields remain unchanged.
### Revision Summary
- After a revision save succeeds, or after a revision resource card is confirmed with no Prompt fields selected for synchronization, output one or two concise plain-text paragraphs. Start with "Updated:" and name only the fields or resources actually changed; state that all other configuration remains unchanged. Do not use ``, a wrapper, a Markdown table, or another card. Even if a revision save emits `agent_generation_completed`, never use the first-generation Completion Summary or claim that a new Agent was generated.
+ After a revision save succeeds, or after a revision resource card is confirmed with no Prompt fields selected for synchronization, compose one or two concise paragraphs, then call `final_answer(revision_summary)` inside one literal `...` block. Start with "Updated:" and name only the fields or resources actually changed; state that all other configuration remains unchanged. Do not call a business tool or wrapper and do not output a Markdown table or another card. Even if a revision save emits `agent_generation_completed`, never use the first-generation Completion Summary or claim that a new Agent was generated.
### Completion Summary
- Only after initial full generation receives `agent_generation_completed`, output the following three concise paragraphs. If the confirmed requirements contain a post-generation scheduling intent, append a fourth paragraph. Do not use ``, a wrapper, a Markdown table, or an interactive card. Only the fourth paragraph may use the specified Markdown link below:
+ Only after initial full generation receives `agent_generation_completed`, compose the following three concise paragraphs and return them with `final_answer(completion_summary)` inside one literal `...` block. If the confirmed requirements contain a post-generation scheduling intent, append a fourth paragraph. Do not call a business tool or wrapper and do not output a Markdown table or interactive card. Only the fourth paragraph may use the specified Markdown link below:
1. State clearly that the new Agent has been generated successfully.
2. Start with "New Agent summary:" and summarize its responsibilities, core capabilities, expected result, and any explicitly abandoned scope from confirmed requirements and real bound resources. Never show raw Prompt content or claim unbound capabilities.
@@ -211,8 +211,8 @@ system_prompt: |-
4. Only for a post-generation scheduling intent, state clearly that this workflow has not created a scheduled task and that it must be created after Agent generation from a conversation with the new Agent. Briefly restate any confirmed time or recurrence information, then direct the user to "open [Scheduled tasks](/agent-tasks), select 'Create in chat,' choose the new Agent, and submit the scheduling request." Never claim that the scheduled task already exists.
### Tool And Termination Rules
- - Except for the Completion Summary, Revision Summary, and revision boundary guidance, use simple valid Python inside literal `` and `` tags.
- - Except for those plain-text outputs, call one business tool per action with keyword arguments, assign its result, and print it exactly once.
+ - Use simple valid Python inside exactly one literal `` and `` pair for every response that completes the run; Completion Summary, Revision Summary, and revision boundary guidance must call `final_answer(...)` there.
+ - For non-terminal actions, call one business tool per action with keyword arguments, assign its result, and print it exactly once.
- Use only defined values and exact parameter names; do not use `if`, `for`, or repeated identical calls.
- Wait for each real tool result before the next action.
- A wrapper call is the final business action of an interactive-card run. After printing it, call no other tool. The runtime emits the structured payload and stops.
diff --git a/backend/prompts/nl2agent_zh.yaml b/backend/prompts/nl2agent_zh.yaml
index d33a669d2..8384e967e 100644
--- a/backend/prompts/nl2agent_zh.yaml
+++ b/backend/prompts/nl2agent_zh.yaml
@@ -18,7 +18,7 @@ system_prompt: |-
- 只有在 `full_generation` 的当前输入是 `requirement_clarification` 提交 action 时,才可以使用其中的回答继续保存描述。局部字段或资源任务只能使用当前任务明确提供的信息和权威草稿信息。
- 智能体显示名称、生成后的变量名、“生成”“创建”等动词以及常识性的领域能力都不算已确认需求,不得据此自行补全任务、使用者、输出或约束。
- 以下完成标准仅适用于首次完整生成。只有描述已保存、资源需求已安装并绑定或明确放弃、全部 Prompt 字段已保存,并收到 `agent_generation_completed` 状态事件,才算完成配置。
- - `full_generation` 才受“首次完整生成”的完成约束:在收到 `agent_generation_completed` 前,不得输出普通完成说明、模拟执行结果或“已为您生成”“已经创建”等表述;每轮必须通过业务 Tool 或 wrapper 推进完整生成状态。局部任务完成后可以输出局部结果或“已更新”总结,但不得声称 Agent 已完成生成。
+ - `full_generation` 才受“首次完整生成”的完成约束:在收到 `agent_generation_completed` 前,不得输出普通完成说明、模拟执行结果或“已为您生成”“已经创建”等表述;每轮必须通过业务 Tool 或 wrapper 推进完整生成状态。局部任务完成后可以通过 `final_answer(...)` 返回局部结果或“已更新”总结,但不得声称 Agent 已完成生成。
### 任务意图与最小流程
- 先判断用户本轮要完成的任务,再结合当前 `nl2agent_verified_state` 选择最小可执行流程。不要先按“生成模式”或“修订模式”给整轮对话贴标签。
@@ -41,7 +41,7 @@ system_prompt: |-
- 资源任务收到 `suggested_resource_installation` action 后,继续当前资源任务;安装完成后必须重新搜索真实的已安装资源。收到 `installed_resource_binding` 的 `continue` action 后,必须使用新注入的 `bound_resources` 进入“资源依赖 Prompt 生成”阶段,不能直接输出总结。没有任何资源需要绑定时,跳过空绑定卡,并基于权威 `bound_resources` 直接进入 Prompt 生成。
- “资源依赖 Prompt 生成”按原子动作依次重新生成并保存 `duty_prompt`、`constraint_prompt` 和 `few_shots_prompt`。这三个字段必须反映真实绑定资源及其已声明输入;不得沿用会引用旧资源、遗漏新能力或编造调用方式的旧内容。
- 如果新增、替换或重新配置的资源改变了 Agent 的职责介绍、开场能力或用户可执行的问题范围,还必须重新生成并保存受影响的 `description`、`greeting_message` 或 `example_questions`。未受影响的字段保持数据库原值;一次保存只能包含当前已确定要更新的字段。
- - 资源依赖 Prompt 生成完成后,输出“修订总结”,说明资源和实际更新的 Prompt 字段。只有用户明确要求移除资源时,才引导其在右侧表单的工具与技能区域操作。
+ - 资源依赖 Prompt 生成完成后,通过 `final_answer(...)` 返回“修订总结”,说明资源和实际更新的 Prompt 字段。只有用户明确要求移除资源时,才引导其在右侧表单的工具与技能区域操作。
- 不支持通过对话移除资源。用户要求移除时,引导其在右侧表单的工具与技能区域操作。替换资源时可以先新增资源,但必须提示用户在该表单中移除旧资源。
- 除空变量名初始化外,`name`、`display_name`、模型设置、发布状态、版本状态以及六个生成字段之外的其他字段都不能通过 NL2Agent 修改。引导用户在右侧对应表单操作,不得调用保存或资源 Tool。
@@ -61,9 +61,9 @@ system_prompt: |-
2. 在完整生成或资源依赖 Prompt 生成中,最新成功工具结果的 `updated_fields` 是 `["duty_prompt"]`:只生成并保存 `constraint_prompt`。
3. 在完整生成或资源依赖 Prompt 生成中,最新成功工具结果的 `updated_fields` 是 `["constraint_prompt"]`:只生成并保存 `few_shots_prompt`。
4. 在完整生成中,最新成功工具结果的 `updated_fields` 是 `["few_shots_prompt"]`:只生成并同时保存 `greeting_message` 和 `example_questions`。
- 5. 在资源依赖 Prompt 生成中,最新成功工具结果的 `updated_fields` 是 `["few_shots_prompt"]`:只保存已确定受资源变化影响的 `description`、`greeting_message` 或 `example_questions`;没有受影响字段时,直接输出“修订总结”。
- 6. 在完整生成中,最新成功工具结果的 `updated_fields` 是 `["greeting_message", "example_questions"]`,并且包含 `agent_generation_completed` 状态事件:只输出“完成总结”规定的普通文本。
- - 局部字段或对话优化任务保存成功后,如果没有待处理的当前卡片 action,直接输出对应的“修订总结”或局部结果。资源任务必须完成资源依赖 Prompt 生成后才能输出“修订总结”。不得因为某个局部字段或对话优化任务更新了 `duty_prompt`、`constraint_prompt` 或其他字段,就生成或保存其他字段。
+ 5. 在资源依赖 Prompt 生成中,最新成功工具结果的 `updated_fields` 是 `["few_shots_prompt"]`:只保存已确定受资源变化影响的 `description`、`greeting_message` 或 `example_questions`;没有受影响字段时,通过唯一 `...` 动作中的 `final_answer(...)` 返回“修订总结”。
+ 6. 在完整生成中,最新成功工具结果的 `updated_fields` 是 `["greeting_message", "example_questions"]`,并且包含 `agent_generation_completed` 状态事件:通过唯一 `...` 动作中的 `final_answer(...)` 返回“完成总结”。
+ - 局部字段或对话优化任务保存成功后,如果没有待处理的当前卡片 action,通过唯一 `...` 动作中的 `final_answer(...)` 返回对应的“修订总结”或局部结果。资源任务必须完成资源依赖 Prompt 生成后才能返回“修订总结”。不得因为某个局部字段或对话优化任务更新了 `duty_prompt`、`constraint_prompt` 或其他字段,就生成或保存其他字段。
- 不得在任一分支中提及、生成或保存后续分支的字段。保存失败时只能修正并重试当前分支一次。
### 变量名初始化
@@ -86,7 +86,7 @@ system_prompt: |-
6. 收到 `suggested_resource_installation` action 后,原样保留 `installed` 与 `skipped` 结果。若 `installed` 为空,视为用户明确选择不安装任何建议的 Tool 或 Skill:不得继续搜索替代资源,也不得使用空资源列表调用 `{{ wrapper_name }}`;下一个原子动作只生成并保存 `duty_prompt`。否则,使用相同 requirements 重新调用 `{{ installed_tool_name }}`,只相信新返回的真实 `tool_id`/`skill_id`;仍未覆盖时把全部已安装和已跳过 candidate refs 放入 `exclude_refs` 搜索替代资源。没有替代资源时使用澄清卡要求用户明确修改需求、放弃需求或结束,禁止声称未绑定的能力可用。
7. 已安装搜索成功后选择覆盖强匹配需求的最小候选集,总数不超过 {{ max_results }}。已绑定 Tool 仍是普通候选:保留搜索分数,并在相同覆盖规则选中它时继续推荐,让绑定卡展示当前配置供用户确认或修改。选中集合非空时,把未改写的候选传给 `{{ recommend_tool_name }}`,解码其结果,再将该字典和同一 `agent_id` 原样传给 `{{ wrapper_name }}` 的 `installed_resource_binding` subtype;选中集合为空时跳过这两个调用,只生成并保存 `duty_prompt`。
8. 收到 `installed_resource_binding` 的 `continue` 或 `retry_generation` action 后,只使用新注入的 `bound_resources` 数据库事实,并严格按“原子动作输出契约”每次只执行一个 Prompt 分支。
- 9. 最后一批 Prompt 保存成功并收到 `agent_generation_completed` 后,直接输出普通文本完成总结,不再调用任何 Tool 或 wrapper。Tool 出错时只依据 `code` 和 `retryable`,最多重试一次。
+ 9. 最后一批 Prompt 保存成功并收到 `agent_generation_completed` 后,在唯一一对字面量 `...` 中调用 `final_answer(completion_summary)`。不再调用任何业务 Tool 或 wrapper。Tool 出错时只依据 `code` 和 `retryable`,最多重试一次。
### 澄清问题 Schema
每个问题包含稳定的 `question_id`、`question_type`(`single_choice`、`multiple_choice` 或 `text`)、简洁 `title` 和 `required`。选择题包含 `options`,并设置 `allow_other=True`、`other_input_expanded=True`。文本题的主输入已经是开放文本,因此两项都设为 `False`。
@@ -201,10 +201,10 @@ system_prompt: |-
Prompt 保存失败时只修正并重试该批一次;第二次失败后停止且不输出完成总结,已成功保存的字段保持不变。
### 修订总结
- 修订字段保存成功后,或修订资源卡已确认且没有选择同步 Prompt 字段时,输出一到两段简洁普通文本。以“已更新:”开头,只说明实际修改的字段或资源,并说明其他配置保持不变。不得使用 ``、wrapper、Markdown 表格或其他卡片。即使修订保存触发 `agent_generation_completed`,也不得使用首次生成的“完成总结”或声称新智能体已完成生成。
+ 修订字段保存成功后,或修订资源卡已确认且没有选择同步 Prompt 字段时,组织一到两段简洁文本,并在唯一一对字面量 `...` 中调用 `final_answer(revision_summary)`。以“已更新:”开头,只说明实际修改的字段或资源,并说明其他配置保持不变。不得调用业务 Tool、wrapper,也不得输出 Markdown 表格或其他卡片。即使修订保存触发 `agent_generation_completed`,也不得使用首次生成的“完成总结”或声称新智能体已完成生成。
### 完成总结
- 只有首次完整生成收到 `agent_generation_completed` 后,才输出以下三段简洁文本;如果确认需求包含生成后定时意图,再追加第四段。不得使用 ``、wrapper、Markdown 表格或交互卡,第四段只能使用下方指定的 Markdown 链接:
+ 只有首次完整生成收到 `agent_generation_completed` 后,才组织以下三段简洁文本,并在唯一一对字面量 `...` 中调用 `final_answer(completion_summary)`;如果确认需求包含生成后定时意图,再追加第四段。不得调用业务 Tool、wrapper,也不得输出 Markdown 表格或交互卡,第四段只能使用下方指定的 Markdown 链接:
1. 明确说明“新智能体已完成生成”。
2. 以“新智能体总结:”开头,根据已确认需求和真实绑定资源概括职责、核心能力、预期结果,以及已明确放弃的范围(如有);不得展示 Prompt 原文或声称拥有未绑定能力。
@@ -212,8 +212,8 @@ system_prompt: |-
4. 仅当存在生成后定时意图时,明确说明本次尚未创建定时任务,需要在 Agent 生成后通过新 Agent 对话创建。简洁复述用户已确认的时间或周期信息(如有),并提供“前往[定时任务](/agent-tasks),点击‘通过会话创建’,选择新智能体并提交定时执行请求”的引导;不得声称定时任务已经创建。
### Tool 与终止规则
- - 除“完成总结”“修订总结”和修订边界说明外,使用简单有效的 Python,并放在字面量 `` 和 `` 标签中。
- - 除上述普通文本输出外,每个动作只使用关键字参数调用一个业务 Tool,将结果赋值并且只打印一次。
+ - 每个结束当前 run 的响应都必须使用简单有效的 Python,并放在唯一一对字面量 `` 和 `` 标签中;“完成总结”“修订总结”和修订边界说明必须在其中调用 `final_answer(...)`。
+ - 非终止动作每次只使用关键字参数调用一个业务 Tool,将结果赋值并且只打印一次。
- 只使用已定义的值和准确参数名,不使用 `if`、`for` 或相同参数的重复调用。
- 等待每次真实工具结果后再进行下一步。
- wrapper 调用是交互卡 run 的最后一个业务动作。打印结果后不得继续调用 Tool;运行时会发送结构化 payload 并终止本轮。
diff --git a/backend/prompts/skill_creation_complicate_en.yaml b/backend/prompts/skill_creation_complicate_en.yaml
index 05eb90ab9..ef105c2c0 100644
--- a/backend/prompts/skill_creation_complicate_en.yaml
+++ b/backend/prompts/skill_creation_complicate_en.yaml
@@ -3,7 +3,8 @@ system_prompt: |-
## Multi-turn conversation
- - If essential information is missing, ask one concise clarification question and do not emit XML control blocks in that turn.
+ - Every response must start with `` on its own line and end with `` on its own line, with no content outside that envelope.
+ - If essential information is missing, ask one concise clarification question inside `` and do not emit ``, ``, or `` in that turn.
- Use both the conversation history and the current skill snapshot when refining a skill.
{% if target_files %}
- This turn is a targeted file modification. Modify only these files: {{ target_files | join(', ') }}.
@@ -15,7 +16,7 @@ system_prompt: |-
- Emit blocks in the order ``, zero or more ``, then ``.
- Put every XML control tag on a standalone line and do not wrap control blocks in Markdown code fences.
- Never quote or explain XML control tags in clarification, reasoning, or summary text; emit them only as real standalone structure.
- - Start structured output directly with `` without a Markdown code fence or language marker.
+ - After ``, start structured output directly with `` without a Markdown code fence or language marker.
{% endif %}
A skill consists of multiple files, including: core description file (SKILL.md), example documents, script code, and more.
@@ -60,6 +61,7 @@ system_prompt: |-
### Single-File Scenario (SKILL.md Only)
+
---
name: your-skill-name
@@ -77,9 +79,11 @@ system_prompt: |-
Your friendly message to the user, such as skill created, feature highlights, etc.
+
### Multi-File Scenario (SKILL.md + Other Files)
+
---
name: your-skill-name
@@ -105,6 +109,7 @@ system_prompt: |-
Your friendly message to the user, such as skill created, feature highlights, etc.
+
### File Reference Declaration Rules (Important)
diff --git a/backend/prompts/skill_creation_complicate_zh.yaml b/backend/prompts/skill_creation_complicate_zh.yaml
index 3bd971ce8..36bafb30d 100644
--- a/backend/prompts/skill_creation_complicate_zh.yaml
+++ b/backend/prompts/skill_creation_complicate_zh.yaml
@@ -3,7 +3,8 @@ system_prompt: |-
## 多轮对话规则
- - 如果需求缺少关键信息,先提出一个简洁的澄清问题;该轮不要输出 XML 控制块。
+ - 每次响应都必须以独占一行的 `` 开始,并以独占一行的 `` 结束;外层之外不得有任何内容。
+ - 如果需求缺少关键信息,在 `` 内提出一个简洁的澄清问题;该轮不要输出 ``、`` 或 ``。
- 修改技能时同时参考对话历史和当前技能快照。
{% if target_files %}
- 本轮是定向文件修改。只能修改这些文件:{{ target_files | join(', ') }}。
@@ -15,7 +16,7 @@ system_prompt: |-
- 输出顺序固定为 ``、零个或多个 ``、``。
- 所有 XML 控制标签必须独占一行,控制块外不要包裹 Markdown 代码围栏。
- 不要在澄清、思考或总结文本中引用或解释 XML 控制标签;它们只能作为真实结构独占一行输出。
- - 输出结构时直接从 `` 开始,不要添加 Markdown 代码围栏或语言标识。
+ - 输出结构时在 `` 后直接输出 ``,不要添加 Markdown 代码围栏或语言标识。
{% endif %}
技能由多个文件组成,包括:核心描述文件(SKILL.md)、示例文档、脚本代码等。
@@ -60,6 +61,7 @@ system_prompt: |-
### 单文件场景(仅需要 SKILL.md)
+
---
name: your-skill-name
@@ -77,9 +79,11 @@ system_prompt: |-
这里是你对用户的友好说明,如技能已创建、功能亮点等
+
### 多文件场景(需要 SKILL.md + 其他文件)
+
---
name: your-skill-name
@@ -109,6 +113,7 @@ system_prompt: |-
这里是你对用户的友好说明,如技能已创建、功能亮点等
+
### 文件引用声明规则(重要)
diff --git a/backend/prompts/skill_creation_simple_en.yaml b/backend/prompts/skill_creation_simple_en.yaml
index d9949faf6..1abe0bf59 100644
--- a/backend/prompts/skill_creation_simple_en.yaml
+++ b/backend/prompts/skill_creation_simple_en.yaml
@@ -3,7 +3,8 @@ system_prompt: |-
## Multi-turn conversation
- - If essential information is missing, ask one concise clarification question and do not emit XML control blocks in that turn.
+ - Every response must start with `` on its own line and end with `` on its own line, with no content outside that envelope.
+ - If essential information is missing, ask one concise clarification question inside `` and do not emit ``, ``, or `` in that turn.
- Use both the conversation history and the current skill snapshot when refining a skill.
{% if target_files %}
- This turn is a targeted file modification. Modify only these files: {{ target_files | join(', ') }}.
@@ -14,7 +15,7 @@ system_prompt: |-
- When generating or modifying a skill, output the complete latest snapshot rather than a partial patch.
- Put every XML control tag on a standalone line and do not wrap control blocks in Markdown code fences.
- Never quote or explain XML control tags in clarification, reasoning, or summary text; emit them only as real standalone structure.
- - Start structured output directly with `` without a Markdown code fence or language marker.
+ - After ``, start structured output directly with `` without a Markdown code fence or language marker.
- Once a `` block starts, never end the response or switch to `` before emitting `` on its own line.
- Before finishing, verify that the output contains exactly one `` and one matching ``, with `` before ``.
{% endif %}
@@ -56,6 +57,7 @@ system_prompt: |-
### Format Example
+
---
name: your-skill-name
@@ -73,6 +75,7 @@ system_prompt: |-
Your friendly message to the user, such as skill created, feature highlights, etc.
+
## Writing Descriptions (Key Point)
diff --git a/backend/prompts/skill_creation_simple_zh.yaml b/backend/prompts/skill_creation_simple_zh.yaml
index c08534f1f..fed1fe0ae 100644
--- a/backend/prompts/skill_creation_simple_zh.yaml
+++ b/backend/prompts/skill_creation_simple_zh.yaml
@@ -3,7 +3,8 @@ system_prompt: |-
## 多轮对话规则
- - 如果需求缺少关键信息,先提出一个简洁的澄清问题;该轮不要输出 XML 控制块。
+ - 每次响应都必须以独占一行的 `` 开始,并以独占一行的 `` 结束;外层之外不得有任何内容。
+ - 如果需求缺少关键信息,在 `` 内提出一个简洁的澄清问题;该轮不要输出 ``、`` 或 ``。
- 修改技能时同时参考对话历史和当前技能快照。
{% if target_files %}
- 本轮是定向文件修改。只能修改这些文件:{{ target_files | join(', ') }}。
@@ -14,7 +15,7 @@ system_prompt: |-
- 一旦生成或修改技能,必须输出最新的完整快照,不要只输出局部补丁。
- 所有 XML 控制标签必须独占一行,控制块外不要包裹 Markdown 代码围栏。
- 不要在澄清、思考或总结文本中引用或解释 XML 控制标签;它们只能作为真实结构独占一行输出。
- - 输出结构时直接从 `` 开始,不要添加 Markdown 代码围栏或语言标识。
+ - 输出结构时在 `` 后直接输出 ``,不要添加 Markdown 代码围栏或语言标识。
- `` 块一旦开始,就不得在输出 `` 前结束响应或切换到 ``;`` 必须独占一行。
- 输出结束前执行结构自检:必须恰好包含一个 `` 和一个与之配对的 ``,且 `` 必须位于 `` 之前。
{% endif %}
@@ -56,6 +57,7 @@ system_prompt: |-
### 格式示例
+
---
name: your-skill-name
@@ -73,6 +75,7 @@ system_prompt: |-
这里是你对用户的友好说明,如技能已创建、功能亮点等
+
## 编写描述(关键)
diff --git a/backend/prompts/utils/prompt_generate_en.yaml b/backend/prompts/utils/prompt_generate_en.yaml
index 4d62bfe19..1396142c6 100644
--- a/backend/prompts/utils/prompt_generate_en.yaml
+++ b/backend/prompts/utils/prompt_generate_en.yaml
@@ -62,7 +62,7 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
- To distinguish between code execution and displaying user code, use 'code' for executing code and 'code' for displaying code
- Note that executed code is not visible to users. If users need to see the code, use 'code' for displaying code.
- After thinking, when you believe you can answer the user's question, you can generate a final answer directly to the user without generating code and stop the loop.
+ After thinking, when you can answer the user, call `final_answer(...)` inside the single `...` action block. Never return a bare-text final answer.
### Python Code Specifications
1. If it is considered to be code that needs to be executed, use 'code'. If the code does not need to be executed for display only, use 'code', where language_type can be python, java, javascript, etc.;
@@ -96,7 +96,9 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
# After tool execution, the system provides the result in subsequent context: The Oriental Pearl TV Tower is located in Lujiazui, Pudong New Area, Shanghai, China, with a height of 468 meters...
Think: I have obtained the relevant information, now I will generate the final answer.
- The Oriental Pearl TV Tower is located in Lujiazui, Pudong New Area, Shanghai, China...
+
+ final_answer("The Oriental Pearl TV Tower is located in Lujiazui, Pudong New Area, Shanghai, China...")
+
---
@@ -111,7 +113,9 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
# After tool execution, the system provides the result in subsequent context: Trip plan completed: High-speed train G2, departs 8:00, arrives Beijing South Station at 11:30; Hotel near Wangfujing; Recommended attractions: Tiananmen, Forbidden City, Great Wall...
Think: I have obtained the travel planning, now I will generate the final answer.
- Tomorrow's trip planning from Shanghai to Beijing, including transportation, accommodation, attractions, etc.
+
+ final_answer("Tomorrow's trip planning from Shanghai to Beijing, including transportation, accommodation, attractions, etc.")
+
---
@@ -134,14 +138,18 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
# After tool execution, the system provides the result in subsequent context: Based on weather data analysis, today is suitable for outdoor activities. Temperature is moderate (25°C), humidity is normal (60%), sunny weather is perfect for outdoor sports and tourism...
Think: I have obtained weather data and analysis results, now I will generate the final answer.
- Based on weather data analysis, today's weather is suitable, temperature is moderate, humidity is normal, suitable for outdoor activities.
+
+ final_answer("Based on weather data analysis, today's weather is suitable, temperature is moderate, humidity is normal, suitable for outdoor activities.")
+
---
Task 4: "Explain what artificial intelligence is"
Think: I can answer this question directly without calling any tools or assistants.
- Artificial Intelligence (AI) is a branch of computer science dedicated to creating systems that can perform tasks that typically require human intelligence...
+
+ final_answer("Artificial Intelligence (AI) is a branch of computer science dedicated to creating systems that can perform tasks that typically require human intelligence...")
+
---
@@ -163,7 +171,8 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
Think: I have obtained the Python quick sort code, now I will generate the final answer.
- The Python quick sort code is as follows:
+
+ final_answer("""The Python quick sort code is as follows:
def quick_sort(arr):
if len(arr) <= 1:
@@ -173,7 +182,8 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
middle = [x for x in arr if x == pivot]
right = [x for x in arr if x > pivot]
return quick_sort(left) + middle + quick_sort(right)
-
+ """)
+
---
@@ -194,7 +204,9 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
# After tool execution, the system provides the result in subsequent context: [search result 1..., search result 2...]
Think: I have both search results. Now I will synthesize the answer.
- Regarding Python design patterns...; regarding Java design patterns...
+
+ final_answer("Regarding Python design patterns...; regarding Java design patterns...")
+
---
@@ -215,7 +227,9 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
# After tool execution, the system provides the result in subsequent context: {"security": "Found 1 SQL injection vulnerability...", "style": "Function names should use snake_case..."}
Think: I have both the security and style analysis results. Now I will merge them into one report.
- Security: SQL injection vulnerability found... Style: rename functions to snake_case...
+
+ final_answer("Security: SQL injection vulnerability found... Style: rename functions to snake_case...")
+
---
@@ -238,7 +252,9 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
# After tool execution, the system provides the result in subsequent context: [search results..., analysis report...]
Think: I have both the search results and the analysis report. Now I will combine them.
- AI trends 2024: ... Sales quarterly analysis: ...
+
+ final_answer("AI trends 2024: ... Sales quarterly analysis: ...")
+
---
@@ -253,7 +269,9 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
# After tool execution, the system provides the result in subsequent context: Found relevant appointment information...
Think: I have the appointment information. Now I will generate the final answer.
- Based on the retrieval results...
+
+ final_answer("Based on the retrieval results...")
+
---
diff --git a/backend/prompts/utils/prompt_generate_zh.yaml b/backend/prompts/utils/prompt_generate_zh.yaml
index 8be897a82..a97eba8e0 100644
--- a/backend/prompts/utils/prompt_generate_zh.yaml
+++ b/backend/prompts/utils/prompt_generate_zh.yaml
@@ -61,7 +61,7 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
- 考虑到代码执行与展示用户代码的区别,使用'代码'表达运行代码,使用'代码'表达展示代码
- 注意运行的代码不会被用户看到,所以如果用户需要看到代码,你需要使用'代码'表达展示代码。
- 在思考结束后,当Agent认为可以回答用户问题,那么可以不生成代码,直接生成最终回答给到用户并停止循环。
+ 在思考结束后,当Agent认为可以回答用户问题,必须在唯一的 `...` 代码块中调用 `final_answer(...)`;禁止输出裸文本最终回答。
### python代码规范
1. 如果认为是需要执行的代码,使用'代码'格式;如果是不需要执行仅用于展示的代码,使用'代码'格式,其中语言类型例如python、java、javascript等;
@@ -95,7 +95,9 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
# 工具执行后,系统在后续上下文中提供结果:东方明珠广播电视塔位于中国上海市浦东新区陆家嘴,塔高468米,是中国著名的地标建筑之一...
思考:我已经获得了有关信息,现在我将生成最终回答。
- 东方明珠广播电视塔位于中国上海市浦东新区陆家嘴...
+
+ final_answer("东方明珠广播电视塔位于中国上海市浦东新区陆家嘴...")
+
---
@@ -110,7 +112,9 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
# 工具执行后,系统在后续上下文中提供结果:行程规划已完成,包括:高铁G2,8:00出发,11:30到达北京南站;酒店预订于王府井附近;景点推荐:天安门、故宫、长城...
思考:我已经获得了出行规划,现在我将生成最终回答。
- 明天从上海出发去北京的行程规划,包括交通、住宿、景点等。
+
+ final_answer("明天从上海出发去北京的行程规划,包括交通、住宿、景点等。")
+
---
@@ -133,14 +137,18 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
# 工具执行后,系统在后续上下文中提供结果:根据天气数据分析,今天天气适宜外出活动,温度适中(25℃),湿度正常(60%),晴天适合户外运动和旅游...
思考:我已经获得了天气数据和分析结果,现在我将生成最终回答。
- 根据天气数据分析,今天天气适宜,温度适中,湿度正常,适合户外活动。
+
+ final_answer("根据天气数据分析,今天天气适宜,温度适中,湿度正常,适合户外活动。")
+
---
任务4:"解释什么是人工智能"
思考:这个问题我可以直接回答,不需要调用任何工具或助手。
- 人工智能(AI)是计算机科学的一个分支,致力于创建能够执行通常需要人类智能的任务的系统...
+
+ final_answer("人工智能(AI)是计算机科学的一个分支,致力于创建能够执行通常需要人类智能的任务的系统...")
+
---
@@ -160,7 +168,8 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
思考:我已经获得了快速排序的python代码,现在我将生成最终回答。
- 快速排序的python代码如下:
+
+ final_answer("""快速排序的python代码如下:
def quick_sort(arr):
if len(arr) <= 1:
@@ -170,7 +179,8 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
middle = [x for x in arr if x == pivot]
right = [x for x in arr if x > pivot]
return quick_sort(left) + middle + quick_sort(right)
-
+ """)
+
---
@@ -191,7 +201,9 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
# 工具执行后,系统在后续上下文中提供结果:[搜索结果1..., 搜索结果2...]
思考:已获得两个搜索结果,现在整合回答。
- Python设计模式方面...,Java设计模式方面...
+
+ final_answer("Python设计模式方面...,Java设计模式方面...")
+
---
@@ -212,7 +224,9 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
# 工具执行后,系统在后续上下文中提供结果:{"security": "发现1个SQL注入漏洞...", "style": "函数名应使用snake_case..."}
思考:已获得安全分析和风格检查结果,现在整合两份报告。
- 安全方面发现SQL注入漏洞...,风格方面建议函数名改为snake_case...
+
+ final_answer("安全方面发现SQL注入漏洞...,风格方面建议函数名改为snake_case...")
+
---
@@ -235,7 +249,9 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
# 工具执行后,系统在后续上下文中提供结果:[搜索结果..., 数据分析报告...]
思考:已获得搜索结果和分析报告,整合回答。
- 2024年AI发展趋势...,销售数据季度趋势分析...
+
+ final_answer("2024年AI发展趋势...,销售数据季度趋势分析...")
+
---
@@ -250,7 +266,9 @@ FEW_SHOTS_SYSTEM_PROMPT: |-
# 工具执行后,系统在后续上下文中提供结果:找到相关出诊信息...
思考:已获得出诊信息,现在我将生成最终回答。
- 根据检索结果...
+
+ final_answer("根据检索结果...")
+
---
diff --git a/backend/utils/content_classifier_utils.py b/backend/utils/content_classifier_utils.py
index 7907655a5..92d0a0a9d 100644
--- a/backend/utils/content_classifier_utils.py
+++ b/backend/utils/content_classifier_utils.py
@@ -31,6 +31,8 @@ def __init__(self):
self._origin_type: Optional[str] = None
self._state_before_file = "others"
self._known_tags = {
+ "",
+ "",
"",
"",
"",
@@ -103,7 +105,11 @@ def _process_tag_start(self, final: bool = False) -> Optional[List[Dict[str, Any
content_after_tag = self.buffer[gt_pos + 1:]
if not content_after_tag and not final:
return None
- if content_after_tag and not content_after_tag.startswith(("\n", "\r\n")):
+ if (
+ content_after_tag
+ and not content_after_tag.startswith(("\n", "\r\n"))
+ and matched not in {"", ""}
+ ):
return self._emit_potential_tag_start()
results.extend(self._handle_matched_tag(gt_pos, potential_tag, matched))
elif len(potential_tag) > self.MAX_TAG_LENGTH:
@@ -258,6 +264,10 @@ def _create_event(self, content: str) -> Dict[str, Any]:
def _handle_tag(self, tag: str) -> Optional[Dict[str, Any]]:
"""Handle matched tag and update state."""
+ if tag in {"", ""}:
+ self.saw_control_tag = True
+ return None
+
if tag == "":
self.saw_control_tag = True
self.state = "skill_body"
diff --git a/backend/utils/context_utils.py b/backend/utils/context_utils.py
index 26cfdcdb6..840d902bf 100644
--- a/backend/utils/context_utils.py
+++ b/backend/utils/context_utils.py
@@ -169,7 +169,7 @@ def _build_execution_flow_text(
lines.append(" - 如果自验证提示存在错误、证据不足、参数不完整或结果不可靠,必须优先修正、补充证据、重新调用工具,或清晰说明无法完成的部分。")
lines.append(" - 最终回答只有在自验证通过后才会展示给用户;如果系统返回 Verification feedback,请根据该反馈继续修正,不要忽略。")
lines.append("")
- lines.append("在思考结束后,当你认为可以回答用户问题,那么可以不生成代码,直接生成最终回答给到用户并停止循环。")
+ lines.append("在思考结束后,当你认为可以回答用户问题,必须在唯一的 `...` 代码块中调用 `final_answer(...)`;禁止输出裸文本最终回答。")
lines.append("")
lines.append("生成最终回答时,你需要遵循以下规范:")
lines.append("1. Markdown格式要求:")
@@ -241,7 +241,7 @@ def _build_execution_flow_text(
lines.append(" - If verification reports errors, insufficient evidence, incomplete parameters, or unreliable results, you must repair the issue, gather more evidence, call tools again, or clearly state what cannot be completed.")
lines.append(" - The final answer is shown to the user only after verification passes. If the system returns Verification feedback, continue revising based on that feedback.")
lines.append("")
- lines.append("After thinking, when you believe you can answer the user's question, you can generate a final answer directly to the user without generating code and stop the loop.")
+ lines.append("After thinking, when you can answer the user, call `final_answer(...)` inside the single `...` action block. Never return a bare-text final answer.")
lines.append("")
lines.append("When generating the final answer, you need to follow these specifications:")
lines.append("1. **Markdown Format Requirements**:")
diff --git a/doc/docs/en/user-guide/start-chat.md b/doc/docs/en/user-guide/start-chat.md
index 7f9aefa65..fa562b73a 100644
--- a/doc/docs/en/user-guide/start-chat.md
+++ b/doc/docs/en/user-guide/start-chat.md
@@ -206,7 +206,7 @@ Nexent agents are implemented using the CodeAgent from [smolagents](https://gith

-The loop repeats the reasoning process in the front end until the model determines that it can generate the final answer directly or the maximum number of steps is reached. The final answer is output in Markdown format and supports headings, lists, tables, code blocks, and links. When retrieval tools are used, citation markers such as `[[letter+number]]` must be added after the relevant content to support traceability.
+In the front end, the reasoning loop repeats until the model explicitly submits a final answer through the Agent Runtime's `final_answer(...)` action or the maximum number of steps is reached. The final answer is output in Markdown format and supports headings, lists, tables, code blocks, and links. When retrieval tools are used, citation markers such as `[[letter+number]]` must be added after the relevant content to support traceability.
### 3. View Code and Tool Calls
diff --git a/doc/docs/zh/user-guide/start-chat.md b/doc/docs/zh/user-guide/start-chat.md
index 2e3eae992..06de0755f 100644
--- a/doc/docs/zh/user-guide/start-chat.md
+++ b/doc/docs/zh/user-guide/start-chat.md
@@ -208,7 +208,7 @@ Nexent 智能体基于 [smolagents](https://github.com/huggingface/smolagents)

-循环逻辑在前端是Reasoning重复直到模型判断可以直接生成最终答案,或达到最大步骤数。最终答案以 Markdown 格式输出,支持标题、列表、表格、代码块和链接;若使用了检索工具,还需在对应内容后添加引用标记 `[[字母+数字]]`,以支持溯源。
+循环逻辑在前端表现为 Reasoning 重复,直到模型通过 Agent Runtime 的 `final_answer(...)` 动作显式提交最终答案,或达到最大步骤数。最终答案以 Markdown 格式输出,支持标题、列表、表格、代码块和链接;若使用了检索工具,还需在对应内容后添加引用标记 `[[字母+数字]]`,以支持溯源。
### 3. 查看代码和工具调用过程
diff --git a/frontend/app/[locale]/agents/components/agentConfig/SkillBuildModal.tsx b/frontend/app/[locale]/agents/components/agentConfig/SkillBuildModal.tsx
index 89c9e75a8..b02b13fd0 100644
--- a/frontend/app/[locale]/agents/components/agentConfig/SkillBuildModal.tsx
+++ b/frontend/app/[locale]/agents/components/agentConfig/SkillBuildModal.tsx
@@ -682,6 +682,13 @@ export default function SkillBuildModal({
if (event.type === "agent_new_run" || event.type === "step_count") {
setIsStreaming(true);
}
+ if (
+ event.type === "model_attempt_control" &&
+ event.phase === "rollback"
+ ) {
+ rollbackDraftStream();
+ return;
+ }
if (event.type === "skill_body" || event.type === "file_content") {
beginDraftStream();
setIsStreaming(true);
diff --git a/frontend/app/[locale]/newchat/adapter/remote-chat-model-adapter.ts b/frontend/app/[locale]/newchat/adapter/remote-chat-model-adapter.ts
index fcc13ce29..f394ec5a0 100644
--- a/frontend/app/[locale]/newchat/adapter/remote-chat-model-adapter.ts
+++ b/frontend/app/[locale]/newchat/adapter/remote-chat-model-adapter.ts
@@ -1701,6 +1701,14 @@ export const remoteChatModelAdapter: ChatModelAdapter = {
const parentReasoning = createReasoningAccumulator(contentParts);
const nl2SkillFilePartIndices = new Map();
let nl2SkillSummaryPartIndex: number | null = null;
+ type Nl2SkillAttemptCheckpoint = {
+ files: Map;
+ summary: { index: number; part: any } | null;
+ };
+ const nl2SkillAttemptCheckpoints = new Map<
+ string,
+ Nl2SkillAttemptCheckpoint
+ >();
const classifyNl2SkillFile = (
path: string
): Pick => {
@@ -1892,6 +1900,66 @@ export const remoteChatModelAdapter: ChatModelAdapter = {
}
}
};
+ const beginNl2SkillAttempt = (attemptId: string) => {
+ if (!isNl2Skill) return;
+ const files = new Map();
+ for (const [path, index] of nl2SkillFilePartIndices) {
+ const part = contentParts[index];
+ files.set(path, {
+ index,
+ part: {
+ ...part,
+ data:
+ part?.data && typeof part.data === "object"
+ ? { ...part.data }
+ : part?.data,
+ },
+ });
+ }
+ const summary =
+ nl2SkillSummaryPartIndex === null
+ ? null
+ : {
+ index: nl2SkillSummaryPartIndex,
+ part: { ...contentParts[nl2SkillSummaryPartIndex] },
+ };
+ nl2SkillAttemptCheckpoints.set(attemptId, { files, summary });
+ };
+ const resolveNl2SkillAttempt = (
+ attemptId: string,
+ phase: "rollback" | "commit"
+ ) => {
+ if (!isNl2Skill) return;
+ const checkpoint = nl2SkillAttemptCheckpoints.get(attemptId);
+ nl2SkillAttemptCheckpoints.delete(attemptId);
+ if (phase !== "rollback" || !checkpoint) return;
+
+ const createdIndices = new Set();
+ for (const [path, index] of nl2SkillFilePartIndices) {
+ if (!checkpoint.files.has(path)) createdIndices.add(index);
+ }
+ if (
+ checkpoint.summary === null &&
+ nl2SkillSummaryPartIndex !== null
+ ) {
+ createdIndices.add(nl2SkillSummaryPartIndex);
+ }
+ for (const index of [...createdIndices].sort((a, b) => b - a)) {
+ removeContentPart(index);
+ }
+
+ nl2SkillFilePartIndices.clear();
+ for (const [path, snapshot] of checkpoint.files) {
+ contentParts[snapshot.index] = snapshot.part;
+ nl2SkillFilePartIndices.set(path, snapshot.index);
+ }
+ if (checkpoint.summary) {
+ contentParts[checkpoint.summary.index] = checkpoint.summary.part;
+ nl2SkillSummaryPartIndex = checkpoint.summary.index;
+ } else {
+ nl2SkillSummaryPartIndex = null;
+ }
+ };
const handleModelAttemptControl = (chunk: SseChunk): boolean => {
if (
chunk.type !== "model_attempt_control" ||
@@ -1904,11 +1972,15 @@ export const remoteChatModelAdapter: ChatModelAdapter = {
if (!top) {
if (chunk.phase === "begin") {
parentReasoning.beginAttempt(chunk.attempt_id);
+ beginNl2SkillAttempt(chunk.attempt_id);
} else if (chunk.phase === "rollback") {
+ resolveNl2SkillAttempt(chunk.attempt_id, "rollback");
parentReasoning.rollbackAttempt(chunk.attempt_id);
} else {
+ resolveNl2SkillAttempt(chunk.attempt_id, "commit");
parentReasoning.commitAttempt(chunk.attempt_id);
}
+ if (isNl2Skill) custom?.onNl2SkillEvent?.(chunk);
return true;
}
diff --git a/frontend/tests/nl2SkillAttemptRollback.test.ts b/frontend/tests/nl2SkillAttemptRollback.test.ts
new file mode 100644
index 000000000..6a84bb99f
--- /dev/null
+++ b/frontend/tests/nl2SkillAttemptRollback.test.ts
@@ -0,0 +1,26 @@
+import assert from "node:assert/strict";
+import { readFile } from "node:fs/promises";
+import test from "node:test";
+
+const read = (path: string) => readFile(new URL(path, import.meta.url), "utf8");
+
+test("NL2Skill rolls rejected model attempts out of both chat and draft UI", async () => {
+ const [adapter, modal] = await Promise.all([
+ read("../app/[locale]/newchat/adapter/remote-chat-model-adapter.ts"),
+ read("../app/[locale]/agents/components/agentConfig/SkillBuildModal.tsx"),
+ ]);
+
+ assert.match(adapter, /beginNl2SkillAttempt\(chunk\.attempt_id\)/);
+ assert.match(
+ adapter,
+ /resolveNl2SkillAttempt\(chunk\.attempt_id, "rollback"\)/
+ );
+ assert.match(adapter, /custom\?\.onNl2SkillEvent\?\.\(chunk\)/);
+ assert.match(modal, /event\.type === "model_attempt_control"/);
+ assert.match(modal, /event\.phase === "rollback"/);
+ assert.match(modal, /rollbackDraftStream\(\)/);
+ assert.doesNotMatch(
+ modal,
+ /event\.type === "model_attempt_control"[\s\S]{0,160}message\.error/
+ );
+});
diff --git a/sdk/benchmark/generic/configs/gsm8k_solver_assistant.yaml b/sdk/benchmark/generic/configs/gsm8k_solver_assistant.yaml
index 0dddd3e9f..6674c39f7 100644
--- a/sdk/benchmark/generic/configs/gsm8k_solver_assistant.yaml
+++ b/sdk/benchmark/generic/configs/gsm8k_solver_assistant.yaml
@@ -139,7 +139,7 @@ prompt_template:
- 注意运行的代码不会被用户看到,所以如果用户需要看到代码,你需要使用'代码'表达展示代码。
- **重要**:代码执行后,系统会返回 "Observation:" 标记的内容(这是真实的执行结果)。请基于这些真实结果继续下一步思考,**不要在代码执行前自行编造观察结果**。
- 在思考结束后,当Agent认为可以回答用户问题,那么可以不生成代码,直接生成最终回答给到用户并停止循环。
+ 在思考结束后,当Agent认为可以回答用户问题,必须在唯一的 `...` 代码块中调用 `final_answer(...)`;禁止输出裸文本最终回答。
### python代码规范
1. 如果认为是需要执行的代码,使用'代码'格式;如果是不需要执行仅用于展示的代码,使用'代码'格式,其中语言类型例如python、java、javascript等;
diff --git a/sdk/nexent/core/agents/agent_model.py b/sdk/nexent/core/agents/agent_model.py
index 3566941f5..f55c436a6 100644
--- a/sdk/nexent/core/agents/agent_model.py
+++ b/sdk/nexent/core/agents/agent_model.py
@@ -256,6 +256,10 @@ class AgentConfig(BaseModel):
ge=1,
)
model_name: str = Field(description="Model alias from ModelConfig")
+ output_protocol: Literal["code_action", "final_answer_envelope"] = Field(
+ description="Closed model-output protocol used by the Agent runtime",
+ default="code_action",
+ )
provide_run_summary: Optional[bool] = Field(
description="Whether to provide run summary to upper-level Agent", default=False
)
diff --git a/sdk/nexent/core/agents/core_agent.py b/sdk/nexent/core/agents/core_agent.py
index 086be61be..95c3af49e 100644
--- a/sdk/nexent/core/agents/core_agent.py
+++ b/sdk/nexent/core/agents/core_agent.py
@@ -1,5 +1,5 @@
-import json
import ast
+import json
import logging
import os
import re
@@ -25,7 +25,7 @@
from ...monitor import get_monitoring_manager
-from ..model_errors import ModelInvocationTerminalError
+from ..model_errors import ModelErrorCode, ModelInvocationTerminalError
from ..utils.observer import MessageObserver, ProcessType
from jinja2 import Template, StrictUndefined
@@ -41,6 +41,17 @@
render_guardrail_refusal,
render_tool_input_refusal,
)
+from .output_protocol import (
+ ExecutableAction,
+ ExplicitFinalAnswer,
+ ModelOutputProtocolError,
+ ProtocolErrorReason,
+ RuntimeFinalAnswer,
+ classify_model_output,
+ has_meaningful_visible_content,
+ unicode_category_summary,
+)
+from .context.budget import message_role
from ..utils.token_estimation import msg_token_count
from .plan_repo import PlanRepo
from ..human_interaction.contracts import AttemptSuspended, RecoveryRequired, RunTerminated, StepSteered
@@ -210,87 +221,6 @@ def convert_code_format(text):
return text
-class FinalAnswerError(Exception):
- """Raised when agent output directly."""
- pass
-
-
-class InvalidActionFormatError(AgentExecutionError):
- """Raised when model output resembles an action but is not executable."""
-
-
-_ACTION_RECORD_LINE_RE = re.compile(
- r"(?im)^\s*(?:[-*#>]\s*)*(?:step\s+\d+\s*:|called\s+tool\b|observation\s*:|tool_calls?\s*:)"
-)
-_ACTION_JSON_KEYS = frozenset({"action", "tool_call", "tool_calls", "arguments"})
-_ACTION_INTENT_RE = re.compile(
- r"(?is)(?:^|\n)\s*(?:(?:思考|分析|thoughts?|analysis)\s*[::].{0,800}"
- r"|(?:我(?:将|需要|先)|接下来|下一步|i\s+(?:will|need\s+to|should)\b|next\b).{0,240})"
- r"(?:调用|使用|检索|搜索|call|use|search|invoke)"
-)
-_EXPLICIT_FINAL_ANSWER_RE = re.compile(
- r"(?is)(?:^|\n)\s*(?:最终回答|final\s+answer)\s*[::]\s*\S"
-)
-
-
-def _looks_like_invalid_action_output(text: Any) -> bool:
- """Return whether non-executable output appears to be an action protocol record."""
- if not isinstance(text, str):
- return False
- stripped = text.strip()
- if not stripped:
- return False
- if _ACTION_RECORD_LINE_RE.search(stripped):
- return True
- if "" in stripped or "" in stripped or "```" in stripped:
- return True
- if stripped.startswith("```") and stripped.endswith("```"):
- first_newline = stripped.find("\n")
- if first_newline != -1:
- stripped = stripped[first_newline + 1:-3].strip()
- if stripped.startswith(("{", "[")):
- try:
- payload = json.loads(stripped)
- except (TypeError, ValueError):
- return False
- records = payload if isinstance(payload, list) else [payload]
- return any(
- isinstance(record, dict) and bool(_ACTION_JSON_KEYS.intersection(record))
- for record in records
- )
- return False
-
-
-def _looks_like_incomplete_action_output(
- text: Any,
- available_tool_names: Any = (),
- finish_reason: Optional[str] = None,
-) -> bool:
- """Identify a truncated or unfinished action that must not become a final answer.
-
- Providers omit the matched stop sequence from their response. A model may
- therefore return only a preamble such as "思考:我将调用
- knowledge_base_search" before the executable ```` block. Treating
- that preamble as a final answer ends the loop after one model call.
- """
- if not isinstance(text, str) or not text.strip():
- return False
- if finish_reason == "length":
- return True
- if _looks_like_invalid_action_output(text):
- return True
- if _EXPLICIT_FINAL_ANSWER_RE.search(text):
- return False
-
- normalized = text.casefold()
- mentioned_tool = any(
- str(tool_name).casefold() in normalized
- for tool_name in available_tool_names or ()
- if tool_name
- )
- return mentioned_tool and bool(_ACTION_INTENT_RE.search(text))
-
-
class ToolInputBlockedError(AgentExecutionError):
"""Raised by the guardrail tool-input wrap when a call is blocked.
@@ -327,7 +257,7 @@ def _screened_tool_forward(engine, tool_name, controller, logger, original_forwa
if action != "pass":
controller.emit(decision.verification_result, message=decision.message)
if action in ("block", "terminate"):
- # Stash the refusal; _step_stream raises FinalAnswerError from it (no retry loop).
+ # Stash the refusal; _step_stream raises a trusted runtime final from it.
refusal = render_tool_input_refusal(decision, tool_name)
controller.pending_tool_block_refusal = refusal
raise ToolInputBlockedError(refusal, logger)
@@ -507,6 +437,10 @@ def __init__(
self.conversation_id = kwargs.pop("conversation_id", None)
self.user_id = kwargs.pop("user_id", None)
self.workspace_path = kwargs.pop("workspace_path", None)
+ self.output_protocol = kwargs.pop("output_protocol", "code_action")
+ if self.output_protocol not in ("code_action", "final_answer_envelope"):
+ raise ValueError(f"Unsupported output protocol: {self.output_protocol}")
+ self._consecutive_protocol_errors = 0
self.human_interaction = None
context_runtime = kwargs.pop("context_runtime", None)
@@ -741,16 +675,25 @@ def _finalize_failed_verification_candidate(
)
action_step.is_final_answer = True
action_step.action_output = controlled_answer
+ self._record_output_protocol(
+ "runtime_final_answer",
+ final_answer_source="final_verifier_controlled_failure",
+ )
return True, controlled_answer
- def _log_model_call_parameters(self, input_messages: List[ChatMessage], stop_sequences: List[str], additional_args: Dict[str, Any]) -> None:
+ def _log_model_call_parameters(
+ self,
+ input_messages: List[ChatMessage],
+ stop_sequences: Optional[List[str]],
+ additional_args: Dict[str, Any],
+ ) -> None:
"""
Log model call parameters with content truncation for readability.
Args:
input_messages: List of chat messages being sent to the model
- stop_sequences: Stop sequences for the model
+ stop_sequences: Optional stop sequences for the model
additional_args: Additional arguments passed to the model
"""
try:
@@ -817,6 +760,99 @@ def _provider_overflow_recovery_safe(self) -> bool:
for step in self.memory.steps[self._history_step_count:]
)
+ def _record_output_protocol(
+ self,
+ classification: str,
+ *,
+ reason: str = "",
+ final_answer_source: str = "",
+ ) -> None:
+ """Attach content-free output-protocol diagnostics to the active trace."""
+
+ consecutive_errors = getattr(self, "_consecutive_protocol_errors", 0)
+ repair_ordinal = consecutive_errors if 0 < consecutive_errors <= 2 else 0
+ attributes = {
+ "agent.output_protocol": getattr(self, "output_protocol", "code_action"),
+ "agent.model_output_classification": classification,
+ "agent.protocol_error_reason": reason,
+ "agent.consecutive_protocol_errors": consecutive_errors,
+ "agent.protocol_repair_ordinal": repair_ordinal,
+ "agent.final_answer_source": final_answer_source,
+ }
+ monitoring_manager = get_monitoring_manager()
+ monitoring_manager.set_span_attributes(**attributes)
+ monitoring_manager.add_span_event("agent.output_protocol", attributes)
+
+ def _controlled_protocol_failure(self) -> str:
+ if str(getattr(self, "lang", "en")).lower().startswith("zh"):
+ return "模型连续未遵循 Agent 输出协议,本次运行已安全终止。请重试或更换模型。"
+ return (
+ "The model repeatedly failed to follow the Agent output protocol, "
+ "so this run was stopped safely. Please retry or use another model."
+ )
+
+ def _resolve_deferred_model_attempt(
+ self,
+ message: ChatMessage | None,
+ *,
+ accepted: bool,
+ ) -> None:
+ """Commit or roll back one successfully streamed model attempt."""
+
+ if message is None or not getattr(message, "model_attempt_commit_deferred", False):
+ return
+ attempt_id = getattr(message, "model_attempt_id", None)
+ attempt_number = getattr(message, "model_attempt_number", None)
+ if not isinstance(attempt_id, str) or not isinstance(attempt_number, int):
+ return
+ method_name = "commit_model_attempt" if accepted else "rollback_model_attempt"
+ resolve = getattr(self.observer, method_name, None)
+ if callable(resolve):
+ resolve(attempt_id, attempt_number)
+ message.model_attempt_commit_deferred = False
+
+ def _append_protocol_repair_context(self, protocol_error: ModelOutputProtocolError) -> None:
+ """Add a safe model-only correction without persisting rejected output."""
+
+ repair_messages = getattr(self, "_protocol_repair_messages", None)
+ if repair_messages is None:
+ repair_messages = []
+ self._protocol_repair_messages = repair_messages
+ repair_messages.append(
+ ChatMessage(
+ role=MessageRole.USER,
+ content=[{
+ "type": "text",
+ "text": protocol_error.repair_instruction,
+ }],
+ )
+ )
+
+ def _ensure_open_model_turn(self, input_messages: list[Any]) -> list[Any]:
+ """End the request with an explicit continuation turn when history ends in assistant."""
+
+ if not input_messages or message_role(input_messages[-1]) != "assistant":
+ return input_messages
+ if getattr(self, "output_protocol", "code_action") == "final_answer_envelope":
+ instruction = (
+ "Continue the current task from the read-only completed-action record above. "
+ "Do not repeat any completed action. Return the next response using the required "
+ "Agent protocol; when complete, return exactly one envelope."
+ )
+ else:
+ instruction = (
+ "Continue the current task from the read-only completed-action record above. "
+ "Do not repeat any completed action. Return exactly one next executable action "
+ "using the required Agent protocol; call final_answer(...) when the task is complete."
+ )
+ return [
+ *input_messages,
+ ChatMessage(
+ role=MessageRole.USER,
+ content=[{"type": "text", "text": instruction}],
+ ),
+ ]
+
def _step_stream(self, memory_step: ActionStep) -> Generator[Any]:
"""
Perform one step in the ReAct framework: the agent thinks, acts, and observes the result.
@@ -826,9 +862,6 @@ def _step_stream(self, memory_step: ActionStep) -> Generator[Any]:
if hitl is not None and memory_step.model_output is not None:
model_output = memory_step.model_output
else:
- self.observer.add_message(
- self.agent_name, ProcessType.STEP_COUNT, self.step_number)
-
final_context = self.context_runtime.prepare_step(
model=self.model,
memory=self.memory,
@@ -850,12 +883,25 @@ def _step_stream(self, memory_step: ActionStep) -> Generator[Any]:
self._last_uncompressed_est = msg_token_count(input_messages, chars_per_token)
# Add new step in logs
memory_step.model_input_messages = input_messages
- stop_sequences = ["Observation:", "Calling tools:"]
+ # Closed output protocols must receive the complete generation.
+ # Legacy CodeAgent stop strings can match a reasoning model's first
+ # tokens; providers then strip the match and return an apparently
+ # successful empty stream (finish_reason=stop). Strict classification
+ # below already rejects fabricated observations and protocol prefixes.
+ stop_sequences: list[str] | None = None
# Prepare additional arguments
additional_args: dict[str, Any] = {}
if self._use_structured_outputs_internally:
additional_args["response_format"] = CODEAGENT_RESPONSE_FORMAT
+ if getattr(self.model, "supports_deferred_attempt_commit", False) is True:
+ additional_args["_defer_attempt_commit"] = True
+
+ repair_messages = getattr(self, "_protocol_repair_messages", [])
+ if repair_messages:
+ input_messages = [*input_messages, *repair_messages]
+ input_messages = self._ensure_open_model_turn(input_messages)
+ memory_step.model_input_messages = input_messages
# Log model call parameters before execution
self._log_model_call_parameters(input_messages, stop_sequences, additional_args)
@@ -871,11 +917,9 @@ def _step_stream(self, memory_step: ActionStep) -> Generator[Any]:
)
if decision.effective_action == "terminate":
self._append_verification_feedback(memory_step, decision.verification_result)
- # Pre-built refusal as the final answer; FinalAnswerError ends the run (no retry loop).
- memory_step.model_output = render_guardrail_refusal(
- decision, input_messages
- )
- raise FinalAnswerError()
+ refusal = render_guardrail_refusal(decision, input_messages)
+ memory_step.model_output = refusal
+ raise RuntimeFinalAnswer(refusal, "guardrail_input")
if decision.effective_action == "mask" and decision.masked_messages is not None:
input_messages = decision.masked_messages
self._append_verification_feedback(memory_step, decision.verification_result)
@@ -908,10 +952,13 @@ def rebuild_after_provider_overflow():
model_output = chat_message.content
memory_step.token_usage = chat_message.token_usage
memory_step.model_output = model_output
-
- self.logger.log_markdown(
- content=model_output, title="MODEL OUTPUT", level=LogLevel.INFO)
- except ModelInvocationTerminalError:
+ except ModelInvocationTerminalError as terminal_error:
+ if terminal_error.error_code == ModelErrorCode.EMPTY_RESPONSE_EXHAUSTED:
+ raise ModelOutputProtocolError(
+ reason=ProtocolErrorReason.EMPTY_VISIBLE_CONTENT,
+ protocol=getattr(self, "output_protocol", "code_action"),
+ logger=self.logger,
+ ) from terminal_error
raise
except Exception as e:
if self.stop_event.is_set():
@@ -919,23 +966,57 @@ def rebuild_after_provider_overflow():
raise AgentGenerationError(
f"Error in generating model output:\n{e}", self.logger) from e
- self.logger.log_markdown(
- content=model_output, title="Output message of the LLM:", level=LogLevel.DEBUG)
+ self.logger.log(
+ "Model output received "
+ f"(length={len(str(model_output or ''))}, "
+ f"unicode_categories={unicode_category_summary(model_output)})",
+ level=LogLevel.DEBUG,
+ )
if hitl is not None:
hitl.generated(memory_step)
- # Parse
+ # Parse using the configured closed output protocol.
try:
if self._use_structured_outputs_internally:
code_action = json.loads(model_output)["code"]
code_action = extract_code_from_text(code_action, self.code_block_tags) or code_action
+ classified_output = ExecutableAction(code=str(code_action))
else:
- code_action = parse_code_blobs(model_output)
+ finish_reason = getattr(self.model, "last_finish_reason", None)
+ if finish_reason is None:
+ diagnostics = getattr(self.model, "last_response_diagnostics", None) or {}
+ finish_reason = diagnostics.get("finish_reason")
+ classified_output = classify_model_output(
+ model_output,
+ protocol=getattr(self, "output_protocol", "code_action"),
+ finish_reason=finish_reason,
+ logger=self.logger,
+ )
+ if isinstance(classified_output, ExplicitFinalAnswer):
+ self._resolve_deferred_model_attempt(memory_step.model_output_message, accepted=True)
+ getattr(self, "_protocol_repair_messages", []).clear()
+ self._consecutive_protocol_errors = 0
+ self._record_output_protocol("explicit_final_answer")
+ self.observer.add_message(
+ self.agent_name, ProcessType.STEP_COUNT, self.step_number)
+ memory_step.action_output = classified_output.answer
+ yield ActionOutput(output=classified_output.answer, is_final_answer=True)
+ return
+
+ code_action = classified_output.code
code_action = fix_final_answer_code(code_action)
code_action = _remove_parallel_executor_import(code_action)
memory_step.code_action = code_action
+ self._resolve_deferred_model_attempt(memory_step.model_output_message, accepted=True)
+ getattr(self, "_protocol_repair_messages", []).clear()
+ self._consecutive_protocol_errors = 0
+ self._record_output_protocol(
+ "legacy_executable_action" if classified_output.legacy_format else "executable_action"
+ )
# Record parsing results
+ self.observer.add_message(
+ self.agent_name, ProcessType.STEP_COUNT, self.step_number)
self.observer.add_message(
self.agent_name, ProcessType.PARSE, code_action)
verification_controller = getattr(self, "verification_controller", None)
@@ -952,32 +1033,18 @@ def rebuild_after_provider_overflow():
self.logger,
)
+ except ModelOutputProtocolError:
+ self._resolve_deferred_model_attempt(memory_step.model_output_message, accepted=False)
+ raise
except AgentExecutionError:
raise
except Exception:
- if _looks_like_incomplete_action_output(
- model_output,
- available_tool_names=self._known_tool_names(),
- finish_reason=getattr(self.model, "last_finish_reason", None),
- ):
- raise InvalidActionFormatError(
- "The previous response described an action but ended before producing an executable tool call. "
- "Do not treat an action preamble as the final answer. Emit executable Python inside "
- "..., or return a complete user-facing final answer.",
- self.logger,
- )
- # Guard: if the model returned empty or whitespace-only content,
- # treat it as a generation error so the retry loop can recover,
- # instead of silently terminating the conversation with no output.
- if not model_output or not str(model_output).strip():
- raise AgentGenerationError(
- "Model returned empty or whitespace-only output; "
- "this is likely a transient API issue and the step will be retried.",
- self.logger,
- )
- self.logger.log_markdown(
- content=model_output, title="AGENT FINAL ANSWER", level=LogLevel.INFO)
- raise FinalAnswerError()
+ self._resolve_deferred_model_attempt(memory_step.model_output_message, accepted=False)
+ raise ModelOutputProtocolError(
+ reason=ProtocolErrorReason.MALFORMED_ACTION,
+ protocol=self.output_protocol,
+ logger=self.logger,
+ )
tool_call = ToolCall(
name="python_interpreter",
@@ -1035,7 +1102,7 @@ def rebuild_after_provider_overflow():
refusal = pending_refusal or getattr(e, "refusal", "")
self.verification_controller.pending_tool_block_refusal = None
memory_step.model_output = refusal
- raise FinalAnswerError()
+ raise RuntimeFinalAnswer(refusal, "guardrail_tool_input")
exec_duration_ms = (time.time() - exec_start) * 1000
if hasattr(self.python_executor, "state") and "_print_outputs" in self.python_executor.state:
execution_logs = str(
@@ -1361,11 +1428,17 @@ def _run_stream(
action_step = None
hitl = getattr(self, "human_interaction", None)
if hitl is not None and hitl.restored and hitl.completed_output is not None:
+ self._record_output_protocol(
+ "runtime_final_answer",
+ final_answer_source="hitl_restored",
+ )
yield FinalAnswerStep(handle_agent_output_types(hitl.completed_output))
return
if hitl is None or not hitl.restored:
self.step_number = 1
returned_final_answer = False
+ self._consecutive_protocol_errors = 0
+ self._protocol_repair_messages: list[ChatMessage] = []
final_verification_round = hitl.final_verification_round if hitl is not None else 0
verification_config = getattr(
self,
@@ -1401,7 +1474,7 @@ def _run_stream(
if isinstance(output, ActionOutput) and output.is_final_answer:
candidate_answer = output.output
- if candidate_answer is None or not str(candidate_answer).strip():
+ if not has_meaningful_visible_content(candidate_answer):
diagnostics = getattr(self.model, "last_response_diagnostics", None)
logger.warning(
"event=empty_final_answer_candidate source=final_answer_tool "
@@ -1409,10 +1482,10 @@ def _run_stream(
self.step_number,
diagnostics,
)
- raise AgentExecutionError(
- "The final_answer tool returned empty content. Call final_answer again "
- "with a non-empty user-facing response.",
- self.logger,
+ raise ModelOutputProtocolError(
+ ProtocolErrorReason.EMPTY_VISIBLE_CONTENT,
+ getattr(self, "output_protocol", "code_action"),
+ logger=self.logger,
)
self.logger.log(
Text(f"Final answer: {candidate_answer}", style=f"bold {YELLOW_HEX}"),
@@ -1433,6 +1506,15 @@ def _run_stream(
self._validate_final_answer(final_answer)
returned_final_answer = True
action_step.is_final_answer = True
+ self._record_output_protocol(
+ "explicit_final_answer",
+ final_answer_source=(
+ "final_answer_envelope"
+ if getattr(self, "output_protocol", "code_action")
+ == "final_answer_envelope"
+ else "final_answer_tool"
+ ),
+ )
else:
returned_final_answer, final_answer = self._finalize_failed_verification_candidate(
action_step=action_step,
@@ -1447,54 +1529,77 @@ def _run_stream(
self._validate_final_answer(final_answer)
returned_final_answer = True
action_step.is_final_answer = True
+ self._record_output_protocol(
+ "explicit_final_answer",
+ final_answer_source=(
+ "final_answer_envelope"
+ if getattr(self, "output_protocol", "code_action")
+ == "final_answer_envelope"
+ else "final_answer_tool"
+ ),
+ )
- except FinalAnswerError:
- # When the model does not output code, directly treat the large model content as the final answer
- candidate_answer = action_step.model_output
- if isinstance(candidate_answer, str):
- candidate_answer = convert_code_format(candidate_answer)
- if candidate_answer is None or not str(candidate_answer).strip():
- diagnostics = getattr(self.model, "last_response_diagnostics", None)
- logger.warning(
- "event=empty_final_answer_candidate source=direct_model_output "
- "step_number=%s model_diagnostics=%s",
- self.step_number,
- diagnostics,
- )
- action_step.error = AgentGenerationError(
- "Model returned empty content instead of a final answer; the step will be retried.",
- self.logger,
- )
- continue
+ except RuntimeFinalAnswer as terminal:
+ final_answer = terminal.answer
+ returned_final_answer = True
+ action_step.is_final_answer = True
+ action_step.action_output = final_answer
+ self._record_output_protocol(
+ "runtime_final_answer",
+ final_answer_source=terminal.source,
+ )
- if verification_config.enabled and verification_config.final_verification_enabled:
- final_verification_round += 1
- verification_result = self.verification_controller.verify_final_answer(
- task=task,
- candidate=candidate_answer,
- memory_summary=self._build_verification_memory_summary(action_step),
- round_number=final_verification_round,
- )
- if verification_result.passed:
- final_answer = candidate_answer
- if self.final_answer_checks:
- self._validate_final_answer(final_answer)
- returned_final_answer = True
- action_step.is_final_answer = True
- else:
- returned_final_answer, final_answer = self._finalize_failed_verification_candidate(
- action_step=action_step,
- verification_result=verification_result,
- verification_round=final_verification_round,
- max_rounds=max_final_verification_rounds,
- candidate_answer=candidate_answer,
- )
- else:
- final_answer = candidate_answer
+ except ModelOutputProtocolError as protocol_error:
+ self._consecutive_protocol_errors += 1
+ self._record_output_protocol(
+ "protocol_error",
+ reason=protocol_error.reason.value,
+ )
+ logger.warning(
+ "event=model_output_protocol_error protocol=%s reason=%s step_number=%s "
+ "consecutive_errors=%s finish_reason=%s unicode_categories=%s",
+ getattr(self, "output_protocol", "code_action"),
+ protocol_error.reason.value,
+ self.step_number,
+ self._consecutive_protocol_errors,
+ getattr(getattr(self, "model", None), "last_finish_reason", None),
+ unicode_category_summary(getattr(action_step, "model_output", None)),
+ )
+ action_has_executed = bool(getattr(action_step, "tool_calls", None))
+ if action_has_executed:
+ # Preserve completed tool evidence so a repair generation cannot
+ # replay an external side effect. The model receives the error
+ # through memory, but the UI warning remains suppressed.
+ action_step.error = protocol_error
+ action_step._suppress_user_error = True
+ elif self._consecutive_protocol_errors >= 3:
+ self._protocol_repair_messages.clear()
+ action_step.model_output = None
+ action_step.model_output_message = None
+ final_answer = self._controlled_protocol_failure()
returned_final_answer = True
action_step.is_final_answer = True
+ action_step.action_output = final_answer
+ self._record_output_protocol(
+ "controlled_protocol_failure",
+ reason=protocol_error.reason.value,
+ final_answer_source="protocol_error_limit",
+ )
+ else:
+ self._append_protocol_repair_context(protocol_error)
+ action_step.model_output = None
+ action_step.model_output_message = None
+ action_step.token_usage = None
+ interrupted = True
+ if hitl is not None:
+ hitl.completed_step(final_verification_round, None)
+ continue
except StepSteered:
+ self._resolve_deferred_model_attempt(
+ getattr(action_step, "model_output_message", None),
+ accepted=False,
+ )
interrupted = True
continue
except ModelInvocationTerminalError:
@@ -1505,6 +1610,10 @@ def _run_stream(
interrupted = True
raise
except (AttemptSuspended, RecoveryRequired, RunTerminated):
+ self._resolve_deferred_model_attempt(
+ getattr(action_step, "model_output_message", None),
+ accepted=False,
+ )
interrupted = True
raise
except AgentError as e:
@@ -1527,6 +1636,10 @@ def _run_stream(
if self.stop_event.is_set():
final_answer = ""
+ self._record_output_protocol(
+ "runtime_final_answer",
+ final_answer_source="user_stop",
+ )
if not returned_final_answer and self.step_number == max_steps + 1:
max_steps_data = json.dumps({
@@ -1539,19 +1652,12 @@ def _run_stream(
# _handle_max_steps_reached already yields the final step internally
# and sets action_step.error, so don't yield again to avoid duplicate error
final_answer = self._handle_max_steps_reached(task)
- if verification_config.enabled and verification_config.final_verification_enabled:
- final_verification_round += 1
- verification_result = self.verification_controller.verify_final_answer(
- task=task,
- candidate=final_answer,
- memory_summary=self._build_verification_memory_summary(),
- round_number=final_verification_round,
- )
- if not verification_result.passed:
- final_answer = self.verification_controller.build_controlled_failure_answer(
- final_answer,
- verification_result,
- )
+ if not has_meaningful_visible_content(final_answer):
+ final_answer = self._controlled_protocol_failure()
+ self._record_output_protocol(
+ "runtime_final_answer",
+ final_answer_source="max_steps",
+ )
if hitl is not None:
hitl.complete_run(final_answer)
yield FinalAnswerStep(handle_agent_output_types(final_answer))
@@ -1724,7 +1830,7 @@ def rebuild_final_after_provider_overflow():
# Guard: if the model returned empty content at max-steps, provide a
# meaningful fallback instead of an empty final_answer.
- if not model_output or not str(model_output).strip():
+ if not has_meaningful_visible_content(model_output):
model_output = (
"The agent was unable to generate a valid response after reaching "
"the maximum number of steps. Please try rephrasing your request."
diff --git a/sdk/nexent/core/agents/nexent_agent.py b/sdk/nexent/core/agents/nexent_agent.py
index 77746eef0..d4b14894d 100644
--- a/sdk/nexent/core/agents/nexent_agent.py
+++ b/sdk/nexent/core/agents/nexent_agent.py
@@ -912,6 +912,7 @@ def create_single_agent(
user_id=self.user_id,
executor=python_executor,
verification_config=getattr(agent_config, "verification_config", None),
+ output_protocol=getattr(agent_config, "output_protocol", "code_action"),
workspace_path=self.workspace_path,
)
agent.stop_event = self.stop_event
@@ -1143,7 +1144,11 @@ def agent_run_with_observer(
})
observer.add_message("", ProcessType.TOKEN_COUNT, json.dumps(token_data))
- if hasattr(step_log, "error") and step_log.error is not None:
+ if (
+ hasattr(step_log, "error")
+ and step_log.error is not None
+ and not getattr(step_log, "_suppress_user_error", False)
+ ):
# Action-step failures are observations in the ReAct loop:
# the model receives them and can repair/retry on the next
# step. Surface them as warnings so the UI does not imply
diff --git a/sdk/nexent/core/agents/output_protocol.py b/sdk/nexent/core/agents/output_protocol.py
new file mode 100644
index 000000000..9ecb91fba
--- /dev/null
+++ b/sdk/nexent/core/agents/output_protocol.py
@@ -0,0 +1,204 @@
+"""Closed output protocol for Nexent CodeAgent runtimes."""
+
+from __future__ import annotations
+
+import ast
+import re
+import unicodedata
+from dataclasses import dataclass
+from enum import Enum
+from typing import Any, Literal
+
+
+OutputProtocol = Literal["code_action", "final_answer_envelope"]
+
+
+class ProtocolErrorReason(str, Enum):
+ """Stable classifications for invalid model output."""
+
+ EMPTY_VISIBLE_CONTENT = "empty_visible_content"
+ UNSUPPORTED_OR_TAG_ONLY_OUTPUT = "unsupported_or_tag_only_output"
+ MALFORMED_ACTION = "malformed_action"
+ MISSING_EXPLICIT_TERMINATION = "missing_explicit_termination"
+ TRUNCATED_GENERATION = "truncated_generation"
+ INVALID_FINAL_ENVELOPE = "invalid_final_envelope"
+
+
+@dataclass(frozen=True)
+class ExecutableAction:
+ """One validated Python action ready for the executor."""
+
+ code: str
+ legacy_format: bool = False
+
+
+@dataclass(frozen=True)
+class ExplicitFinalAnswer:
+ """One validated NL2Skill final-answer envelope payload."""
+
+ answer: str
+
+
+class ModelOutputProtocolError(Exception):
+ """A recoverable model-output protocol violation."""
+
+ def __init__(
+ self,
+ reason: ProtocolErrorReason,
+ protocol: OutputProtocol,
+ logger: Any = None,
+ ) -> None:
+ self.reason = reason
+ self.protocol = protocol
+ self.repair_instruction = protocol_repair_instruction(protocol, reason)
+ super().__init__(self.repair_instruction)
+ self.message = self.repair_instruction
+
+
+class RuntimeFinalAnswer(Exception):
+ """A trusted runtime-controlled terminal answer."""
+
+ def __init__(self, answer: Any, source: str) -> None:
+ super().__init__(source)
+ self.answer = answer
+ self.source = source
+
+
+_CODE_RE = re.compile(r"\A(?P[\s\S]*)\Z")
+_RUN_RE = re.compile(r"\A```(?P[\s\S]*?)```\Z")
+_FINAL_ENVELOPE_RE = re.compile(r"\A(?P[\s\S]*)\Z")
+_TAG_RE = re.compile(r"?[A-Za-z][^<>]{0,255}>")
+
+
+def _is_protocol_padding(character: str) -> bool:
+ return character.isspace() or unicodedata.category(character) == "Cf"
+
+
+def strip_protocol_padding(value: Any) -> str:
+ """Strip only protocol-ignorable edge whitespace and format characters."""
+
+ text = "" if value is None else str(value)
+ start = 0
+ end = len(text)
+ while start < end and _is_protocol_padding(text[start]):
+ start += 1
+ while end > start and _is_protocol_padding(text[end - 1]):
+ end -= 1
+ return text[start:end]
+
+
+def has_meaningful_visible_content(value: Any) -> bool:
+ """Return whether content contains a non-whitespace, non-format character."""
+
+ if value is None:
+ return False
+ return any(not _is_protocol_padding(character) for character in str(value))
+
+
+def unicode_category_summary(value: Any) -> str:
+ """Return a content-free Unicode category count for diagnostics."""
+
+ counts: dict[str, int] = {}
+ for character in str(value or ""):
+ category = unicodedata.category(character)
+ counts[category] = counts.get(category, 0) + 1
+ return ",".join(f"{key}:{counts[key]}" for key in sorted(counts)) or "empty"
+
+
+def protocol_repair_instruction(
+ protocol: OutputProtocol,
+ reason: ProtocolErrorReason,
+) -> str:
+ """Build safe feedback that teaches only the configured runtime protocol."""
+
+ prefix = f"The previous response violated the Agent output protocol ({reason.value}). "
+ if protocol == "final_answer_envelope":
+ return (
+ prefix + "Return exactly one complete ... envelope. "
+ "Put the required , , and content inside it, with no content outside the envelope."
+ )
+ return (
+ prefix + "Return exactly one executable Python action inside .... "
+ "To finish, call final_answer(...) inside that code block; never return a bare-text final answer."
+ )
+
+
+def _raise_protocol_error(
+ reason: ProtocolErrorReason,
+ protocol: OutputProtocol,
+ logger: Any,
+) -> None:
+ raise ModelOutputProtocolError(reason, protocol, logger)
+
+
+def classify_model_output(
+ output: Any,
+ *,
+ protocol: OutputProtocol,
+ finish_reason: str | None = None,
+ logger: Any = None,
+) -> ExecutableAction | ExplicitFinalAnswer:
+ """Classify one complete model response using a closed runtime protocol."""
+
+ if protocol not in ("code_action", "final_answer_envelope"):
+ raise ValueError(f"Unsupported output protocol: {protocol}")
+ if finish_reason == "length":
+ _raise_protocol_error(ProtocolErrorReason.TRUNCATED_GENERATION, protocol, logger)
+
+ text = strip_protocol_padding(output)
+ if not has_meaningful_visible_content(text):
+ _raise_protocol_error(ProtocolErrorReason.EMPTY_VISIBLE_CONTENT, protocol, logger)
+
+ if protocol == "code_action":
+ code_match = _CODE_RE.fullmatch(text)
+ if code_match:
+ code = code_match.group("body").strip()
+ if not has_meaningful_visible_content(code):
+ _raise_protocol_error(ProtocolErrorReason.MALFORMED_ACTION, protocol, logger)
+ try:
+ ast.parse(code)
+ except (SyntaxError, ValueError, TypeError):
+ _raise_protocol_error(ProtocolErrorReason.MALFORMED_ACTION, protocol, logger)
+ return ExecutableAction(code=code)
+
+ run_match = _RUN_RE.fullmatch(text)
+ if run_match and text.count("```") == 1:
+ code = run_match.group("body").strip()
+ if not has_meaningful_visible_content(code):
+ _raise_protocol_error(ProtocolErrorReason.MALFORMED_ACTION, protocol, logger)
+ try:
+ ast.parse(code)
+ except (SyntaxError, ValueError, TypeError):
+ _raise_protocol_error(ProtocolErrorReason.MALFORMED_ACTION, protocol, logger)
+ return ExecutableAction(code=code, legacy_format=True)
+
+ if any(marker in text for marker in ("", "", "```")):
+ _raise_protocol_error(ProtocolErrorReason.MALFORMED_ACTION, protocol, logger)
+ if _TAG_RE.search(text):
+ _raise_protocol_error(
+ ProtocolErrorReason.UNSUPPORTED_OR_TAG_ONLY_OUTPUT,
+ protocol,
+ logger,
+ )
+ _raise_protocol_error(
+ ProtocolErrorReason.MISSING_EXPLICIT_TERMINATION,
+ protocol,
+ logger,
+ )
+
+ envelope_match = _FINAL_ENVELOPE_RE.fullmatch(text)
+ if envelope_match and text.count("") == 1 and text.count("") == 1:
+ answer = envelope_match.group("body")
+ if not has_meaningful_visible_content(answer):
+ _raise_protocol_error(
+ ProtocolErrorReason.INVALID_FINAL_ENVELOPE,
+ protocol,
+ logger,
+ )
+ return ExplicitFinalAnswer(answer=answer)
+
+ _raise_protocol_error(
+ ProtocolErrorReason.INVALID_FINAL_ENVELOPE,
+ protocol,
+ logger,
+ )
diff --git a/sdk/nexent/core/models/openai_llm.py b/sdk/nexent/core/models/openai_llm.py
index 295d02c6b..1fa719259 100644
--- a/sdk/nexent/core/models/openai_llm.py
+++ b/sdk/nexent/core/models/openai_llm.py
@@ -120,6 +120,8 @@ def _is_timeout_error(exc: BaseException) -> bool:
class OpenAIModel(OpenAIServerModel):
+ supports_deferred_attempt_commit = True
+
# Public SDK constructor: keep common kwargs explicit and read extension
# kwargs below to preserve backward-compatible keyword call sites.
def __init__(self, observer: MessageObserver = MessageObserver, temperature=0.2, top_p=0.95,
@@ -278,6 +280,7 @@ def __call__(self, messages: List[Dict[str, Any]], stop_sequences: Optional[List
_token_tracker=None, context_budget_snapshot: Optional[ContextBudgetSnapshot] = None,
context_rebuild=None, _overflow_recovery_ordinal: int = 0,
_model_attempts_used: int = 0,
+ _defer_attempt_commit: bool = False,
**kwargs, ) -> ChatMessage:
_monitoring_operation.set("chat_completion")
@@ -321,6 +324,7 @@ def __call__(self, messages: List[Dict[str, Any]], stop_sequences: Optional[List
context_rebuild=context_rebuild,
_overflow_recovery_ordinal=_overflow_recovery_ordinal,
_model_attempts_used=_model_attempts_used,
+ _defer_attempt_commit=_defer_attempt_commit,
**kwargs,
)
@@ -721,10 +725,19 @@ def _close_stream_once():
)
message.raw = current_request
message.role = MessageRole.ASSISTANT
- commit_attempt = getattr(self.observer, "commit_model_attempt", None)
- if callable(commit_attempt):
- commit_attempt(attempt_id, attempt)
- self._monitoring.add_span_event("model_attempt_commit", {
+ message.model_attempt_id = attempt_id
+ message.model_attempt_number = attempt
+ message.model_attempt_commit_deferred = _defer_attempt_commit
+ attempt_event = (
+ "model_attempt_commit_deferred"
+ if _defer_attempt_commit
+ else "model_attempt_commit"
+ )
+ if not _defer_attempt_commit:
+ commit_attempt = getattr(self.observer, "commit_model_attempt", None)
+ if callable(commit_attempt):
+ commit_attempt(attempt_id, attempt)
+ self._monitoring.add_span_event(attempt_event, {
"attempt_id": attempt_id,
"attempt": attempt,
})
@@ -839,6 +852,7 @@ def _close_stream_once():
context_rebuild=context_rebuild,
_overflow_recovery_ordinal=_overflow_recovery_ordinal + 1,
_model_attempts_used=attempt,
+ _defer_attempt_commit=_defer_attempt_commit,
**kwargs,
)
is_timeout = _is_timeout_error(e)
diff --git a/sdk/nexent/core/tools/create_scheduled_task_tool.py b/sdk/nexent/core/tools/create_scheduled_task_tool.py
index 230806c63..742d382d4 100644
--- a/sdk/nexent/core/tools/create_scheduled_task_tool.py
+++ b/sdk/nexent/core/tools/create_scheduled_task_tool.py
@@ -28,14 +28,14 @@ class CreateScheduledTaskProposalTool(Tool):
"Create a pending scheduled-task proposal when the user explicitly asks "
"for a task to run later or repeatedly. Pass the user's scheduling request "
"verbatim. This tool only extracts and saves a proposal for user confirmation; "
- "it never executes the business task. Call it as the only action in the code "
- "block and return its result directly as the final answer."
+ "it never executes the business task. Make it the only business-tool call in "
+ "the code block and pass its result to final_answer(...)."
)
description_zh = (
"当用户明确要求未来、延迟或周期性执行任务时,创建一个待确认的"
"定时任务提案。request_text 必须原样传入用户的定时执行请求。"
"此工具只提取并保存待用户确认的提案,不会立即执行业务任务。"
- "调用时它必须是代码块中的唯一动作,并将返回结果直接作为最终回答。"
+ "它必须是代码块中唯一的业务工具调用,并将其结果传给 final_answer(...)。"
)
inputs = {
"request_text": {
diff --git a/test/assets/test_prompt.yaml b/test/assets/test_prompt.yaml
index 38c9debeb..14e6774cd 100644
--- a/test/assets/test_prompt.yaml
+++ b/test/assets/test_prompt.yaml
@@ -21,10 +21,10 @@ system_prompt: |-
- 查看代码执行结果
- 根据结果决定下一步行动
- 在思考结束后,当你认为可以回答用户问题,那么可以不生成代码,直接生成最终回答给到用户并停止循环。
+ 在思考结束后,当你认为可以回答用户问题,必须在唯一的 `...` 代码块中调用 `final_answer(...)`;禁止输出裸文本最终回答。
生成最终回答时,你需要遵顼以下规范:
- 1.不要输出代码,因为最终回答不应该包含任何代码。
+ 1.除非用户要求展示代码,否则 `final_answer(...)` 的正文不应包含代码。
2.使用Markdown格式格式化你的输出。
3.在回答的对应位置添加引用标记,格式为'[[1]][[2]]'。注意仅添加引用标记,不需要添加链接、参考文献等多余内容。
@@ -99,4 +99,4 @@ managed_agent:
task: |-
report: |-
- {{final_answer}}
\ No newline at end of file
+ {{final_answer}}
diff --git a/test/assets/test_sub_prompt.yaml b/test/assets/test_sub_prompt.yaml
index a06b886de..989318088 100644
--- a/test/assets/test_sub_prompt.yaml
+++ b/test/assets/test_sub_prompt.yaml
@@ -19,10 +19,10 @@ system_prompt: |-
3. 观察结果:
- 查看代码执行结果
- 在思考结束后,当你认为可以回答用户问题,那么可以不生成代码,直接生成最终回答给到用户并停止循环。
+ 在思考结束后,当你认为可以回答用户问题,必须在唯一的 `...` 代码块中调用 `final_answer(...)`;禁止输出裸文本最终回答。
生成最终回答时,你需要遵顼以下规范:
- 1.不要输出代码,因为最终回答不应该包含任何代码。
+ 1.除非用户要求展示代码,否则 `final_answer(...)` 的正文不应包含代码。
2.使用Markdown格式格式化你的输出。
3.在回答的对应位置添加引用标记,格式为'[[index]]',其中index为引用的序号。注意仅添加引用标记,不需要添加链接、参考文献等多余内容。
@@ -131,4 +131,4 @@ managed_agent:
即使你的任务解决不成功,也请返回尽可能多的上下文,这样你的管理者可以根据这个反馈采取行动。
report: |-
- {{final_answer}}
\ No newline at end of file
+ {{final_answer}}
diff --git a/test/backend/services/test_nl2skill_service.py b/test/backend/services/test_nl2skill_service.py
index 8f4dada44..42d12ed33 100644
--- a/test/backend/services/test_nl2skill_service.py
+++ b/test/backend/services/test_nl2skill_service.py
@@ -31,6 +31,7 @@ def test_create_nl2skill_agent_config_sets_ephemeral_runtime_options():
assert config.instructions == "system"
assert config.tools == []
assert config.max_steps == 5
+ assert config.output_protocol == "final_answer_envelope"
assert config.provide_run_summary is False
assert config.enable_planning is False
@@ -202,13 +203,14 @@ async def test_stream_preserves_raw_types_and_emits_semantic_events(mocker):
async def fake_agent_run(_run_info, *, thread_manager):
assert thread_manager is not None
chunks = [
- {"type": "model_thinking_output", "content": "Preparing.\n\n\n---\nname: demo\ndescription: Demo\ntags: [demo]\n---\n# Demo\n\n",
},
{"type": "model_output_code", "content": '\nprint("ok")\n\n'},
- {"type": "model_output_thinking", "content": "\nReady.\n\n"},
+ {"type": "model_output_thinking", "content": "\nReady.\n\n"},
{"type": "final_answer", "content": "duplicate"},
]
for chunk in chunks:
@@ -231,6 +233,7 @@ async def fake_agent_run(_run_info, *, thread_manager):
for item in payloads
)
assert any(item["type"] == "summary" for item in payloads)
+ assert not any("FINAL_ANSWER" in item.get("content", "") for item in payloads)
assert not any(item.get("content") == "duplicate" for item in payloads)
assert payloads[-1]["type"] == "done"
assert stop_event.is_set()
diff --git a/test/backend/utils/test_content_classifier_utils.py b/test/backend/utils/test_content_classifier_utils.py
index 182bbfb80..4ba4b1771 100644
--- a/test/backend/utils/test_content_classifier_utils.py
+++ b/test/backend/utils/test_content_classifier_utils.py
@@ -8,6 +8,32 @@
class TestContentClassifier:
"""Test cases for ContentClassifier."""
+ def test_ac_010_final_answer_envelope_is_consumed(self):
+ classifier = ContentClassifier()
+
+ events = classifier.classify(
+ "\n\n# Demo\n\n"
+ "\nCreated.\n\n",
+ origin_type="model_output",
+ )
+ events.extend(classifier.flush())
+
+ assert all("FINAL_ANSWER" not in event.get("content", "") for event in events)
+ assert any(event["type"] == "skill_body" and "# Demo" in event["content"] for event in events)
+ assert any(event["type"] == "summary" and "Created." in event["content"] for event in events)
+ assert classifier.saw_control_tag is True
+
+ def test_ac_010_final_answer_envelope_can_be_adjacent_to_payload(self):
+ classifier = ContentClassifier()
+
+ events = classifier.classify(
+ "clarification?",
+ origin_type="model_output",
+ )
+ events.extend(classifier.flush())
+
+ assert "".join(event.get("content", "") for event in events) == "clarification?"
+
def test_basic_classification(self):
"""Test basic content classification."""
classifier = ContentClassifier()
diff --git a/test/sdk/core/agents/test_agent_model.py b/test/sdk/core/agents/test_agent_model.py
index b9ef26b81..eaa43864a 100644
--- a/test/sdk/core/agents/test_agent_model.py
+++ b/test/sdk/core/agents/test_agent_model.py
@@ -1256,6 +1256,7 @@ def test_agent_config_defaults(self):
)
assert config.prompt_templates is None
assert config.max_steps == 15
+ assert config.output_protocol == "code_action"
assert config.provide_run_summary is False
assert config.instructions is None
assert config.managed_agents == []
diff --git a/test/sdk/core/agents/test_core_agent.py b/test/sdk/core/agents/test_core_agent.py
index 0968b2aaf..e431e5ac6 100644
--- a/test/sdk/core/agents/test_core_agent.py
+++ b/test/sdk/core/agents/test_core_agent.py
@@ -401,51 +401,6 @@ def test_provider_overflow_recovery_is_disabled_after_a_tool_call():
# ----------------------------------------------------------------------------
-def test_incomplete_action_preamble_is_not_a_final_answer():
- output = "思考:我需要先调用 knowledge_base_search 检索当前选择的知识库。"
-
- assert core_agent_module._looks_like_incomplete_action_output(
- output,
- available_tool_names={"knowledge_base_search"},
- ) is True
-
-
-def test_complete_answer_that_names_tool_is_not_misclassified():
- output = "knowledge_base_search 是用于检索知识库的工具。"
-
- assert core_agent_module._looks_like_incomplete_action_output(
- output,
- available_tool_names={"knowledge_base_search"},
- ) is False
-
-
-@pytest.mark.parametrize(
- "output",
- [
- (
- "思考:工具调用成功。根据策略,我需要用 `final_answer` 返回工具结果。\n\n"
- "最终回答:\n定时任务提案已生成,请核对任务内容和执行时间后确认创建。"
- ),
- (
- "Analysis: The tool call succeeded, so I will use `final_answer` to return the result.\n\n"
- "Final answer:\nThe scheduled-task proposal is ready for confirmation."
- ),
- ],
-)
-def test_complete_explicit_final_answer_is_not_misclassified(output):
- assert core_agent_module._looks_like_incomplete_action_output(
- output,
- available_tool_names={"final_answer", "create_scheduled_task_proposal"},
- ) is False
-
-
-def test_length_truncated_non_code_output_is_not_a_final_answer():
- assert core_agent_module._looks_like_incomplete_action_output(
- "这是一个尚未完成的回答",
- finish_reason="length",
- ) is True
-
-
def test_parse_code_blobs_run_format():
"""Test parse_code_blobs with ... pattern (new format)."""
text = """Here is some code:
@@ -1040,62 +995,18 @@ def test_convert_code_format_mixed_with_code():
# ----------------------------------------------------------------------------
-# Tests for FinalAnswerError exception class
+# Tests for trusted runtime final answers
# ----------------------------------------------------------------------------
-def test_final_answer_error_creation():
- """Test FinalAnswerError can be created and raised."""
- error = core_agent_module.FinalAnswerError()
- assert isinstance(error, Exception)
- with pytest.raises(core_agent_module.FinalAnswerError):
+def test_runtime_final_answer_creation():
+ """Trusted runtime final answers carry their content and source."""
+ error = core_agent_module.RuntimeFinalAnswer("refusal", "guardrail_input")
+ assert error.answer == "refusal"
+ assert error.source == "guardrail_input"
+ with pytest.raises(core_agent_module.RuntimeFinalAnswer):
raise error
-@pytest.mark.parametrize(
- "output",
- [
- "Step 2:\nCalled tool 'python_interpreter'()",
- "### Step 2:\n- Called tool 'python_interpreter'()",
- "Observation: previous result",
- '{"tool_calls":[{"name":"python_interpreter","arguments":"print(1)"}]}',
- '```json\n{"action":"search","arguments":{"q":"GAIA"}}\n```',
- "print('missing closing tag')",
- ],
-)
-def test_action_like_non_executable_output_is_not_a_final_answer(output):
- assert core_agent_module._looks_like_invalid_action_output(output) is True
-
-
-@pytest.mark.parametrize(
- "output",
- [
- None,
- 42,
- "",
- " ",
- "The answer is 42.",
- "I could not find enough evidence to answer.",
- '{"answer":"42"}',
- "{not valid json",
- "```json```",
- '["not an action record"]',
- ],
-)
-def test_plain_answer_does_not_look_like_invalid_action(output):
- assert core_agent_module._looks_like_invalid_action_output(output) is False
-
-
-@pytest.mark.parametrize(
- "output",
- [
- "```print('missing closing fence')",
- '[{"action":"search","arguments":{"q":"GAIA"}}]',
- ],
-)
-def test_additional_action_protocol_variants_are_invalid(output):
- assert core_agent_module._looks_like_invalid_action_output(output) is True
-
-
# ----------------------------------------------------------------------------
# Additional edge case tests for parse_code_blobs
# ----------------------------------------------------------------------------
@@ -2318,7 +2229,7 @@ def test_step_stream_uses_context_runtime_for_uncompressed_est(self):
generator = agent._step_stream(action_step)
try:
next(generator)
- except (StopIteration, ValueError):
+ except (StopIteration, ValueError, module.ModelOutputProtocolError):
pass
assert agent._last_uncompressed_est == 5000
@@ -2358,7 +2269,7 @@ def invoke_rebuild(messages, **kwargs):
stream = agent._step_stream(action_step)
try:
list(stream)
- except (ValueError, TypeError):
+ except (ValueError, TypeError, module.ModelOutputProtocolError):
# Parsing the synthetic response is outside this callback contract test.
pass
@@ -2405,7 +2316,7 @@ def test_step_stream_falls_back_without_uncompressed_runtime_count(self):
generator = agent._step_stream(action_step)
try:
next(generator)
- except (StopIteration, ValueError):
+ except (StopIteration, ValueError, module.ModelOutputProtocolError):
pass
# When the runtime has no raw count, fall back to msg_token_count.
@@ -2441,11 +2352,171 @@ def test_step_stream_rejects_whitespace_only_model_output(self, monkeypatch):
action_step = MagicMock()
stream = agent._step_stream(action_step)
- with pytest.raises(Exception, match="empty or whitespace-only output"):
+ with pytest.raises(module.ModelOutputProtocolError) as exc_info:
next(stream)
+ assert exc_info.value.reason == module.ProtocolErrorReason.EMPTY_VISIBLE_CONTENT
assert action_step.model_output == " \n\t"
+ def test_step_stream_turns_exhausted_empty_response_into_silent_protocol_repair(
+ self, monkeypatch
+ ):
+ """Physical empty retries hand off to the hidden semantic repair loop."""
+ module = core_agent_module
+ CoreAgent = module.CoreAgent
+ monkeypatch.setattr(module, "AgentGenerationError", type("AgentGenerationError", (Exception,), {}))
+
+ agent = object.__new__(CoreAgent)
+ agent.agent_name = "test"
+ agent.observer = MagicMock()
+ agent.step_number = 1
+ agent.memory = MagicMock(steps=[])
+ agent.logger = MagicMock()
+ agent.context_runtime = self._context_runtime_mock()
+ final_context = MagicMock()
+ final_context.messages = [MagicMock()]
+ agent.context_runtime.prepare_step.return_value = final_context
+ agent._history_step_count = 0
+ agent._context_tools = MagicMock(return_value=[])
+ agent._use_structured_outputs_internally = False
+ agent.output_protocol = "code_action"
+ agent.verification_controller = None
+ terminal = module.ModelInvocationTerminalError(
+ module.ModelErrorCode.EMPTY_RESPONSE_EXHAUSTED,
+ 5,
+ cause=RuntimeError("provider detail"),
+ )
+ agent.model = MagicMock(side_effect=terminal)
+
+ action_step = MagicMock(model_output=None)
+ with pytest.raises(module.ModelOutputProtocolError) as exc_info:
+ next(agent._step_stream(action_step))
+
+ assert exc_info.value.reason == module.ProtocolErrorReason.EMPTY_VISIBLE_CONTENT
+ assert exc_info.value.__cause__ is terminal
+
+ def test_step_stream_closed_protocol_omits_legacy_provider_stop_sequences(
+ self, monkeypatch
+ ):
+ """Provider-side stops cannot erase a reasoning model's opening prefix."""
+ module = core_agent_module
+ agent = object.__new__(module.CoreAgent)
+ agent.agent_name = "test"
+ agent.observer = MagicMock()
+ agent.step_number = 1
+ agent.memory = MagicMock(steps=[])
+ agent.logger = MagicMock()
+ agent.context_runtime = self._context_runtime_mock()
+ final_context = MagicMock()
+ final_context.messages = [MagicMock()]
+ agent.context_runtime.prepare_step.return_value = final_context
+ agent._history_step_count = 0
+ agent._context_tools = MagicMock(return_value=[])
+ agent._use_structured_outputs_internally = False
+ agent._protocol_repair_messages = []
+ agent.output_protocol = "final_answer_envelope"
+ agent.verification_controller = None
+ response = SimpleNamespace(
+ content="ok",
+ token_usage=None,
+ )
+ agent.model = MagicMock(return_value=response)
+
+ next(agent._step_stream(MagicMock(model_output=None)))
+
+ assert agent.model.call_args.kwargs["stop_sequences"] is None
+
+ def test_step_stream_opens_turn_after_completed_assistant_action(self, monkeypatch):
+ """A completed action cannot leave chat completion ending in assistant."""
+ module = core_agent_module
+ agent = object.__new__(module.CoreAgent)
+ agent.agent_name = "test"
+ agent.observer = MagicMock()
+ agent.step_number = 2
+ agent.memory = MagicMock(steps=[])
+ agent.logger = MagicMock()
+ agent.context_runtime = self._context_runtime_mock()
+ final_context = MagicMock()
+ final_context.messages = [
+ {"role": "user", "content": "task"},
+ {"role": "assistant", "content": "completed action"},
+ ]
+ agent.context_runtime.prepare_step.return_value = final_context
+ agent._history_step_count = 0
+ agent._context_tools = MagicMock(return_value=[])
+ agent._use_structured_outputs_internally = False
+ agent._protocol_repair_messages = []
+ agent.output_protocol = "final_answer_envelope"
+ agent.verification_controller = None
+ agent.model = MagicMock(
+ return_value=SimpleNamespace(
+ content="ok",
+ token_usage=None,
+ )
+ )
+
+ next(agent._step_stream(MagicMock(model_output=None)))
+
+ actual_messages = agent.model.call_args.args[0]
+ assert len(actual_messages) == 3
+ continuation = module.ChatMessage.call_args.kwargs["content"][0]["text"]
+ assert "Do not repeat any completed action" in continuation
+ assert "" in continuation
+
+ def test_step_stream_rolls_back_deferred_attempt_before_protocol_repair(self):
+ """Rejected model text is rolled back before CoreAgent asks for repair."""
+ module = core_agent_module
+ agent = object.__new__(module.CoreAgent)
+ agent.agent_name = "test"
+ agent.observer = MagicMock()
+ agent.step_number = 1
+ agent.memory = MagicMock(steps=[])
+ agent.logger = MagicMock()
+ agent.context_runtime = self._context_runtime_mock()
+ final_context = MagicMock()
+ final_context.messages = [MagicMock()]
+ agent.context_runtime.prepare_step.return_value = final_context
+ agent._history_step_count = 0
+ agent._context_tools = MagicMock(return_value=[])
+ agent._use_structured_outputs_internally = False
+ agent._protocol_repair_messages = []
+ agent.output_protocol = "code_action"
+ agent.verification_controller = None
+
+ response = SimpleNamespace(
+ content='prefix\nfinal_answer("ok")',
+ token_usage=None,
+ model_attempt_id="semantic-attempt",
+ model_attempt_number=1,
+ model_attempt_commit_deferred=True,
+ )
+ model = MagicMock(return_value=response)
+ model.supports_deferred_attempt_commit = True
+ model.last_finish_reason = "stop"
+ model.last_response_diagnostics = {"finish_reason": "stop"}
+ agent.model = model
+
+ action_step = SimpleNamespace(
+ model_output=None,
+ model_output_message=None,
+ token_usage=None,
+ model_input_messages=None,
+ )
+
+ with pytest.raises(module.ModelOutputProtocolError):
+ list(agent._step_stream(action_step))
+
+ assert model.call_args.kwargs["_defer_attempt_commit"] is True
+ agent.observer.rollback_model_attempt.assert_called_once_with(
+ "semantic-attempt", 1
+ )
+ agent.observer.commit_model_attempt.assert_not_called()
+ assert response.model_attempt_commit_deferred is False
+ assert all(
+ call_.args[1] is not module.ProcessType.STEP_COUNT
+ for call_ in agent.observer.add_message.call_args_list
+ )
+
def test_run_stream_stop_event_path_real_execution(self):
"""Test _run_stream with stop_event set (user break)."""
import threading
@@ -2588,95 +2659,23 @@ def mock_step_stream(action_step):
max_steps_calls = [c for c in observer_calls if c[1] == TestProcessType.MAX_STEPS_REACHED]
assert len(max_steps_calls) == 0
- def test_run_stream_final_answer_error_path(self):
- """Test _run_stream when FinalAnswerError is raised."""
- # This covers the code path where the model outputs non-code text (FinalAnswerError)
-
- # Create ProcessType
- class TestProcessType:
- MAX_STEPS_REACHED = "MAX_STEPS_REACHED"
-
- # Track observer calls
- observer_calls = []
-
- # Load CoreAgent
- module = self._load_core_agent_in_isolation()
- CoreAgent = module.CoreAgent
-
- # Verify it's a real class
- assert not isinstance(CoreAgent, MagicMock)
-
- # Get FinalAnswerError from the loaded module
- FinalAnswerError = module.FinalAnswerError
-
- # Create mock memory
- mock_memory = MagicMock()
- mock_memory.steps = []
-
- # Create stop_event not set
- stop_event = MagicMock()
- stop_event.is_set = lambda: False
-
- # Track step_stream calls
- step_stream_calls = [0]
-
- # Create mock ActionStep with model_output
- mock_action_step = MagicMock()
- mock_action_step.model_output = "This is my final answer"
- mock_action_step.is_final_answer = True
-
- # Create step_stream that raises FinalAnswerError
- def mock_step_stream(action_step):
- step_stream_calls[0] += 1
- # Return the mock action step that has model_output
- yield mock_action_step
- # Then raise FinalAnswerError to trigger the except block
- raise FinalAnswerError()
+ def test_run_stream_trusted_runtime_final_path(self, monkeypatch):
+ """A trusted runtime refusal terminates without model-final verification."""
+ module = core_agent_module
+ agent = self._create_canonical_run_agent(monkeypatch)
+ agent.verification_controller = MagicMock()
- # Create agent
- agent = object.__new__(CoreAgent)
- agent.agent_name = "test_agent"
- agent.observer = MagicMock()
- agent.observer.add_message = lambda *args: observer_calls.append(args)
- agent.stop_event = stop_event
- agent.step_number = 1
- agent.memory = mock_memory
- agent.logger = MagicMock()
- agent.logger.log = lambda *args, **kwargs: None
- agent.monitor = MagicMock()
- agent.max_steps = 10
- agent.name = "test_agent"
- agent.task = "test task"
- agent.state = {}
- agent.final_answer_checks = None
- agent.return_full_result = False
- agent.python_executor = MagicMock()
- agent.model = MagicMock()
- agent.prompt_templates = {}
- agent.tools = {}
- agent.managed_agents = {}
- agent.provide_run_summary = False
- agent._use_structured_outputs_internally = False
- agent.context_runtime = self._context_runtime_mock()
- agent.step_metrics = []
+ def mock_step_stream(_action_step):
+ if False:
+ yield None
+ raise module.RuntimeFinalAnswer("refusal", "guardrail_input")
agent._step_stream = mock_step_stream
- agent._handle_max_steps_reached = MagicMock(return_value="Max steps")
- agent._finalize_step = lambda x: None
-
- # Call _run_stream
- generator = agent._run_stream("test task", max_steps=10)
+ results = list(agent._run_stream("test task", max_steps=3))
- # Consume the generator
- try:
- results = list(generator)
- except FinalAnswerError:
- # The generator may raise FinalAnswerError - that's okay
- pass
-
- # FinalAnswerError path should prevent MAX_STEPS_REACHED
- max_steps_calls = [c for c in observer_calls if c[1] == TestProcessType.MAX_STEPS_REACHED]
- assert len(max_steps_calls) == 0
+ assert results[-1].output == "refusal"
+ assert len(agent.memory.steps) == 1
+ agent.verification_controller.verify_final_answer.assert_not_called()
def test_run_stream_retries_empty_final_answer_tool_result(self, monkeypatch):
"""An empty final_answer tool result must not end the run successfully."""
@@ -2712,8 +2711,46 @@ def mock_step_stream(_action_step):
results = list(agent._run_stream("test task", max_steps=2))
assert results[-1].output == "valid answer"
+ assert len(agent.memory.steps) == 1
+ assert getattr(agent.memory.steps[0], "error", None) is None
+ assert agent.memory.steps[0].step_number == 1
+
+ def test_protocol_repair_retains_executed_action_without_user_warning(self, monkeypatch):
+ """A post-execution protocol error keeps tool evidence to prevent replay."""
+ module = core_agent_module
+
+ class FakeAgentError(Exception):
+ pass
+
+ class FakeActionOutput:
+ def __init__(self, output, is_final_answer):
+ self.output = output
+ self.is_final_answer = is_final_answer
+
+ monkeypatch.setattr(module, "AgentError", FakeAgentError)
+ monkeypatch.setattr(module, "ActionOutput", FakeActionOutput)
+ agent = self._create_canonical_run_agent(monkeypatch)
+ calls = 0
+
+ def mock_step_stream(action_step):
+ nonlocal calls
+ calls += 1
+ if calls == 1:
+ action_step.tool_calls = [SimpleNamespace(id="executed")]
+ raise module.ModelOutputProtocolError(
+ module.ProtocolErrorReason.EMPTY_VISIBLE_CONTENT,
+ "code_action",
+ )
+ yield FakeActionOutput(output="done", is_final_answer=True)
+
+ agent._step_stream = mock_step_stream
+ results = list(agent._run_stream("test task", max_steps=2))
+
+ assert results[-1].output == "done"
assert len(agent.memory.steps) == 2
- assert agent.memory.steps[0].error is not None
+ assert agent.memory.steps[0].tool_calls[0].id == "executed"
+ assert agent.memory.steps[0]._suppress_user_error is True
+ assert agent.memory.steps[1].step_number == 2
def test_cmsr_004_terminal_model_error_stops_react_without_memory_append(
self, monkeypatch
@@ -2751,14 +2788,14 @@ def failing_step(_action_step):
agent._finalize_step.assert_not_called()
agent._collect_step_metrics.assert_not_called()
- def test_planning_run_retries_empty_direct_answer_then_verifies_valid_answer(self, monkeypatch):
- """Planning runs reset state, retry an empty answer, and verify the next answer."""
+ def test_planning_run_stops_after_three_protocol_errors(self, monkeypatch):
+ """Planning state resets and three invalid generations fail safely."""
module = core_agent_module
agent = self._create_canonical_run_agent(
monkeypatch,
enable_planning=True,
- model=MagicMock(last_response_diagnostics={"finish_reason": "length"}),
+ model=MagicMock(last_response_diagnostics={"finish_reason": "stop"}),
verification_config=SimpleNamespace(
enabled=True,
final_verification_enabled=True,
@@ -2768,32 +2805,26 @@ def test_planning_run_retries_empty_direct_answer_then_verifies_valid_answer(sel
agent.current_plan = "stale plan"
agent.current_step_index = 99
agent.verification_controller = MagicMock()
- agent.verification_controller.verify_final_answer.return_value = SimpleNamespace(
- passed=True
- )
- agent._build_verification_memory_summary = MagicMock(return_value="summary")
-
- direct_answers = iter([" \n", "valid direct answer"])
-
def mock_step_stream(action_step):
- action_step.model_output = next(direct_answers)
+ action_step.model_output = "bare text"
if False:
yield None
- raise module.FinalAnswerError()
+ raise module.ModelOutputProtocolError(
+ module.ProtocolErrorReason.MISSING_EXPLICIT_TERMINATION,
+ "code_action",
+ )
agent._step_stream = mock_step_stream
- results = list(agent._run_stream("test task", max_steps=2))
+ results = list(agent._run_stream("test task", max_steps=5))
- assert results[-1].output == "valid direct answer"
+ assert "failed to follow" in results[-1].output
assert agent.current_plan is None
assert agent.current_step_index == 0
- assert len(agent.memory.steps) == 2
- assert agent.memory.steps[0].error is not None
- agent.verification_controller.verify_final_answer.assert_called_once()
- assert agent.verification_controller.verify_final_answer.call_args.kwargs[
- "candidate"
- ] == "valid direct answer"
+ assert len(agent.memory.steps) == 1
+ assert getattr(agent.memory.steps[0], "error", None) is None
+ assert agent.memory.steps[0].step_number == 1
+ agent.verification_controller.verify_final_answer.assert_not_called()
# ----------------------------------------------------------------------------
# Tests for _handle_max_steps_reached method
diff --git a/test/sdk/core/agents/test_core_agent_planning.py b/test/sdk/core/agents/test_core_agent_planning.py
index d0f610e7e..ef832e443 100644
--- a/test/sdk/core/agents/test_core_agent_planning.py
+++ b/test/sdk/core/agents/test_core_agent_planning.py
@@ -9,15 +9,11 @@
"""
import importlib.util
-import json
import sys
-import types
from pathlib import Path
from types import ModuleType, SimpleNamespace
from unittest.mock import MagicMock
-import pytest
-
REPO_ROOT = Path(__file__).resolve().parents[4]
@@ -215,6 +211,10 @@ def __init__(self):
token_mod = _sdk_pkg("sdk.nexent.core.utils.token_estimation")
token_mod.msg_token_count = lambda *a, **k: 0
+# context budget helper stub
+budget_mod = _sdk_pkg("sdk.nexent.core.agents.context.budget")
+budget_mod.message_role = lambda message: getattr(message, "role", "")
+
# verification stub
verification_mod = _sdk_pkg("sdk.nexent.core.agents.verification")
@@ -275,13 +275,22 @@ class _VerificationResult:
# ---- Load core_agent under controlled sys.modules -----------------
+OUTPUT_PROTOCOL_PATH = REPO_ROOT / "sdk" / "nexent" / "core" / "agents" / "output_protocol.py"
+OUTPUT_PROTOCOL_NAME = "sdk.nexent.core.agents.output_protocol"
+output_protocol_spec = importlib.util.spec_from_file_location(OUTPUT_PROTOCOL_NAME, OUTPUT_PROTOCOL_PATH)
+output_protocol_module = importlib.util.module_from_spec(output_protocol_spec)
+sys.modules[OUTPUT_PROTOCOL_NAME] = output_protocol_module
+agents_mod = sys.modules["sdk.nexent.core.agents"]
+agents_mod.output_protocol = output_protocol_module
+assert output_protocol_spec and output_protocol_spec.loader
+output_protocol_spec.loader.exec_module(output_protocol_module)
+
CORE_AGENT_PATH = REPO_ROOT / "sdk" / "nexent" / "core" / "agents" / "core_agent.py"
CORE_AGENT_NAME = "sdk.nexent.core.agents.core_agent"
sys.modules["sdk.nexent.core"].__path__ = [str(REPO_ROOT / "sdk" / "nexent" / "core")]
spec = importlib.util.spec_from_file_location(CORE_AGENT_NAME, CORE_AGENT_PATH)
core_agent_module = importlib.util.module_from_spec(spec)
sys.modules[CORE_AGENT_NAME] = core_agent_module
-agents_mod = sys.modules["sdk.nexent.core.agents"]
agents_mod.core_agent = core_agent_module
assert spec and spec.loader
spec.loader.exec_module(core_agent_module)
diff --git a/test/sdk/core/agents/test_guardrail_checkpoints.py b/test/sdk/core/agents/test_guardrail_checkpoints.py
index 4d1bd556b..c7f3afaa5 100644
--- a/test/sdk/core/agents/test_guardrail_checkpoints.py
+++ b/test/sdk/core/agents/test_guardrail_checkpoints.py
@@ -10,8 +10,11 @@
from unittest.mock import MagicMock
import pytest
-
-from nexent.core.agents.agent_model import AgentVerificationConfig, GuardrailConfig, GuardrailRule
+from nexent.core.agents.agent_model import (
+ AgentVerificationConfig,
+ GuardrailConfig,
+ GuardrailRule,
+)
from nexent.core.agents.core_agent import CoreAgent, ToolInputBlockedError
from nexent.core.agents.verification import VerificationController
@@ -182,7 +185,12 @@ def test_guardrail_wrap_tools_no_engine_is_noop():
# ---------------------------------------------------------------------------
import threading as _threading
-from nexent.core.agents.core_agent import FinalAnswerError, InvalidActionFormatError
+
+from nexent.core.agents.output_protocol import (
+ ModelOutputProtocolError,
+ ProtocolErrorReason,
+ RuntimeFinalAnswer,
+)
def _make_step_agent(rule, messages, model_output="ok"):
@@ -214,6 +222,8 @@ def _make_step_agent(rule, messages, model_output="ok"):
agent._last_uncompressed_est = 0
agent._context_tools = MagicMock(return_value=[])
agent._use_structured_outputs_internally = False
+ agent.output_protocol = "code_action"
+ agent._consecutive_protocol_errors = 0
agent._ephemeral_system_messages = None
agent.verification_controller = controller
agent.verification_config = controller.config
@@ -237,7 +247,7 @@ def _msg(role, content):
def test_step_stream_checkpoint1_terminate():
- """Checkpoint ①: block rule on new_input → terminate → FinalAnswerError with refusal."""
+ """Checkpoint ①: blocked input terminates through the trusted runtime path."""
rule = GuardrailRule(
name="destructive_rm",
pattern=r"(?", "bare answer", "final_answer('42')"):
+ response = MagicMock()
+ response.content = content
+ response.token_usage = None
+ responses.append(response)
+ agent.model.side_effect = responses
+ agent.model.last_finish_reason = "stop"
+ agent.enable_planning = False
+ agent.final_answer_checks = None
+ agent.verification_config = AgentVerificationConfig(enabled=False)
+ agent.verification_controller.config.step_verification_enabled = False
+ agent.verification_controller.config.final_verification_enabled = False
+ agent._finalize_step = MagicMock()
+ agent._collect_step_metrics = MagicMock()
+ code_output = MagicMock(output="42", logs="", is_final_answer=True)
+ agent.python_executor.return_value = code_output
+
+ results = list(agent._run_stream("solve this", max_steps=4))
+
+ assert agent.model.call_count == 3
+ assert len(agent.memory.steps) == 1
+ assert agent.memory.steps[0].error is None
+ assert agent.memory.steps[0].step_number == 1
+ assert agent.memory.steps[0].is_final_answer is True
+ final_model_messages = agent.model.call_args_list[-1].args[0]
+ repair_text = "\n".join(
+ str(message.get("content") if isinstance(message, dict) else message.content)
+ for message in final_model_messages
+ )
+ assert "unsupported_or_tag_only_output" in repair_text
+ assert "missing_explicit_termination" in repair_text
+ assert isinstance(results[-1], FinalAnswerStep)
+ assert results[-1].output == "42"
+
+
def test_step_stream_checkpoint2_mask():
"""Checkpoint ②: tool output with keyword → mask → observation redacted."""
rule = GuardrailRule(name="pii", pattern="机密信息", severity="mask")
@@ -359,15 +418,34 @@ def test_step_stream_checkpoint2_mask():
action_step = MagicMock()
try:
next(agent._step_stream(action_step))
- except (FinalAnswerError, StopIteration):
+ except StopIteration:
pass
obs = str(action_step.observations)
assert "机密信息" not in obs
assert "***" in obs
+def test_valid_action_resets_consecutive_protocol_errors():
+ """Any syntactically valid action resets the consecutive protocol-error counter."""
+ rule = GuardrailRule(name="irrelevant", pattern="never-match", severity="block")
+ agent = _make_step_agent(
+ rule,
+ messages=[_msg("user", "hello")],
+ model_output="print(1)",
+ )
+ agent._consecutive_protocol_errors = 2
+ code_output = MagicMock(output="ok", logs="", is_final_answer=False)
+ agent.python_executor.return_value = code_output
+ agent.verification_controller.config.step_verification_enabled = False
+ action_step = MagicMock()
+
+ list(agent._step_stream(action_step))
+
+ assert agent._consecutive_protocol_errors == 0
+
+
def test_step_stream_checkpoint3_except_block():
- """Checkpoint ③: pending_refusal + python_executor raises → FinalAnswerError."""
+ """Checkpoint ③: a stashed refusal raises a trusted runtime final."""
rule = GuardrailRule(
name="destructive_rm",
pattern=r"(?", "🙂", "\u200bA"])
+def test_ac_001_visible_content_is_meaningful(value):
+ assert has_meaningful_visible_content(value) is True
+
+
+@pytest.mark.parametrize(
+ ("output", "reason"),
+ [
+ ("\u200b\u2060\ufeff", ProtocolErrorReason.EMPTY_VISIBLE_CONTENT),
+ ("unsupported", ProtocolErrorReason.UNSUPPORTED_OR_TAG_ONLY_OUTPUT),
+ ("", ProtocolErrorReason.UNSUPPORTED_OR_TAG_ONLY_OUTPUT),
+ ("plain final answer", ProtocolErrorReason.MISSING_EXPLICIT_TERMINATION),
+ ("print(1)", ProtocolErrorReason.MALFORMED_ACTION),
+ (
+ "print(1)print(2)",
+ ProtocolErrorReason.MALFORMED_ACTION,
+ ),
+ ("prefixprint(1)", ProtocolErrorReason.MALFORMED_ACTION),
+ ],
+)
+def test_ac_001_ac_002_invalid_code_outputs_are_protocol_errors(output, reason):
+ with pytest.raises(ModelOutputProtocolError) as exc_info:
+ classify_model_output(output, protocol="code_action")
+
+ assert exc_info.value.reason == reason
+
+
+def test_ac_008_length_finish_reason_is_never_executable_or_final():
+ with pytest.raises(ModelOutputProtocolError) as exc_info:
+ classify_model_output(
+ "final_answer('partial')",
+ protocol="code_action",
+ finish_reason="length",
+ )
+
+ assert exc_info.value.reason == ProtocolErrorReason.TRUNCATED_GENERATION
+
+
+def test_ac_003_exact_code_action_is_executable():
+ result = classify_model_output(
+ "\u200b\nfinal_answer('done')\ufeff",
+ protocol="code_action",
+ )
+
+ assert result == ExecutableAction(code="final_answer('done')")
+
+
+def test_ac_004_code_action_preserves_protocol_like_text_inside_final_answer():
+ code = '''final_answer("""Markdown: ```python\nprint(1)\n```\nHTML: x\nXML: 值🙂""")'''
+
+ result = classify_model_output(
+ f"{code}",
+ protocol="code_action",
+ )
+
+ assert result == ExecutableAction(code=code)
+
+
+def test_legacy_run_action_remains_executable_only():
+ result = classify_model_output(
+ "```\nprint('legacy')\n```",
+ protocol="code_action",
+ )
+
+ assert result == ExecutableAction(code="print('legacy')", legacy_format=True)
+
+
+def test_ac_004_final_envelope_preserves_arbitrary_payload():
+ payload = "\n# Skill\n```python\nprint('')\n```\n🙂\n"
+ result = classify_model_output(
+ f"{payload}",
+ protocol="final_answer_envelope",
+ )
+
+ assert result == ExplicitFinalAnswer(answer=payload)
+
+
+@pytest.mark.parametrize(
+ "output",
+ [
+ "",
+ "outsideinside",
+ "onetwo",
+ "nested",
+ "unfinished",
+ ],
+)
+def test_ac_010_invalid_final_envelopes_are_rejected(output):
+ with pytest.raises(ModelOutputProtocolError) as exc_info:
+ classify_model_output(output, protocol="final_answer_envelope")
+
+ assert exc_info.value.reason == ProtocolErrorReason.INVALID_FINAL_ENVELOPE
diff --git a/test/sdk/core/models/test_openai_llm.py b/test/sdk/core/models/test_openai_llm.py
index 8aa9cc247..74510ad8c 100644
--- a/test/sdk/core/models/test_openai_llm.py
+++ b/test/sdk/core/models/test_openai_llm.py
@@ -2238,6 +2238,34 @@ def test_call_without_tracker_creates_tracker(openai_model_instance):
mock_tracker.record_token.assert_called()
+def test_call_can_defer_successful_attempt_commit_for_core_agent(openai_model_instance):
+ """CoreAgent may validate a successful stream before committing it to clients."""
+ mock_chunk = MagicMock()
+ mock_chunk.choices = [MagicMock()]
+ mock_chunk.choices[0].delta.content = 'final_answer("ok")'
+ mock_chunk.choices[0].delta.role = "assistant"
+ mock_chunk.choices[0].delta.reasoning = None
+ mock_chunk.choices[0].delta.reasoning_content = None
+ mock_chunk.choices[0].finish_reason = "stop"
+ mock_chunk.usage = MagicMock(prompt_tokens=1, completion_tokens=1)
+ openai_model_instance.observer.reset_mock()
+
+ with patch.object(openai_model_instance, "_prepare_completion_kwargs", return_value={}):
+ openai_model_instance.client.chat.completions.create.return_value = [mock_chunk]
+ result = openai_model_instance(
+ messages=[{"role": "user", "content": "hello"}],
+ _token_tracker=MagicMock(),
+ _defer_attempt_commit=True,
+ )
+
+ openai_model_instance.observer.begin_model_attempt.assert_called_once()
+ openai_model_instance.observer.commit_model_attempt.assert_not_called()
+ openai_model_instance.observer.rollback_model_attempt.assert_not_called()
+ assert result.model_attempt_commit_deferred is True
+ assert isinstance(result.model_attempt_id, str)
+ assert result.model_attempt_number == 1
+
+
def test_call_token_estimation_with_list_content(openai_model_instance):
"""Test __call__ method extracts text from list-formatted content when usage info is None (line 220)."""