Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions backend/app/services/agent_runtime/node_executor.py
Original file line number Diff line number Diff line change
Expand Up @@ -778,6 +778,8 @@ async def _model(
repair_limit = (
WRITE_FILE_PROTOCOL_REPAIR_LIMIT
if is_write_file_repair
else 10
if repair_code == "invalid_tool_call"
else 1
)
repair_counter_key = (
Expand Down
20 changes: 15 additions & 5 deletions backend/app/services/agent_runtime/tool_execution.py
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,7 @@

from app.models.agent_run import AgentRun
from app.models.agent_tool_execution import AgentToolExecution
from app.services.builtin_tool_definitions import BUILTIN_TOOL_NAMES

ToolExecutionStatus = Literal[
"not_started",
Expand All @@ -44,7 +45,7 @@
"reconcile",
]
ToolSideEffectState = Literal["none", "confirmed", "possible", "unknown"]
SAFE_READ_MAX_ATTEMPTS = 3
SAFE_READ_MAX_ATTEMPTS = 10

# These tools dispatch an external image-generation request and can therefore
# leave the provider outcome uncertain after a response timeout. Direct Chat
Expand Down Expand Up @@ -2011,7 +2012,7 @@ async def reconcile_unknown_tool_execution(
if not is_user_reconcilable_unknown_execution(execution):
raise ToolExecutionError(
"tool_execution_reconciliation_not_supported",
"manual reconciliation is only supported for conditional write_file or image-generation receipts",
"manual reconciliation is not supported for this Tool receipt",
)

prior_metadata = (
Expand Down Expand Up @@ -2078,12 +2079,21 @@ def is_user_reconcilable_unknown_execution(execution: AgentToolExecution) -> boo
new tool call, so the original provider request is never replayed.
"""
effect, retry_policy = _execution_metadata(execution)
contract_version = getattr(execution, "contract_version", None)
tool_name = str(getattr(execution, "tool_name", "") or "")
is_registered_dynamic_mcp = (
tool_name not in BUILTIN_TOOL_NAMES
and isinstance(contract_version, str)
and contract_version.startswith(f"registered:{tool_name}:")
and effect == "external_write"
and retry_policy == "never"
)
return (
execution.tool_name == "write_file"
tool_name == "write_file"
and effect == "write"
and retry_policy == "conditional"
) or (
execution.tool_name in _IMAGE_GENERATION_TOOL_NAMES
tool_name in _IMAGE_GENERATION_TOOL_NAMES
and effect == "external_write"
and retry_policy == "never"
)
) or is_registered_dynamic_mcp
2 changes: 1 addition & 1 deletion backend/app/services/agent_runtime/tool_repair_budget.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,7 +10,7 @@
from app.services.agent_runtime.state import JsonObject

SAME_FINGERPRINT_FAILURE_LIMIT = 10
TOOL_EPISODE_FAILURE_LIMIT = 20
TOOL_EPISODE_FAILURE_LIMIT = 10
_REPAIRABLE_MODEL_ACTIONS = frozenset(
{"repair_arguments", "choose_other_tool"}
)
Expand Down
4 changes: 2 additions & 2 deletions backend/app/services/llm/caller.py
Original file line number Diff line number Diff line change
Expand Up @@ -66,7 +66,7 @@ async def execute_tool(*args, **kwargs):
"send_message_to_agent", "send_feishu_message", "send_email"
})

WRITE_FILE_PROTOCOL_REPAIR_LIMIT = 3
WRITE_FILE_PROTOCOL_REPAIR_LIMIT = 10
WRITE_FILE_PROTOCOL_REPAIR_COUNTER_KEY = "invalid_tool_call:write_file"
WRITE_FILE_PROTOCOL_REPAIR_INSTRUCTION = (
"Your previous `write_file` call was not executed because `function.arguments` "
Expand Down Expand Up @@ -788,7 +788,7 @@ async def _buffer_chunk(_text: str) -> None:
repair_limit = (
WRITE_FILE_PROTOCOL_REPAIR_LIMIT
if retry_tool_name == "write_file"
else 1
else 10
)
repair_counter_key = (
WRITE_FILE_PROTOCOL_REPAIR_COUNTER_KEY
Expand Down
2 changes: 1 addition & 1 deletion backend/tests/test_agent_runtime_model_step_service.py
Original file line number Diff line number Diff line change
Expand Up @@ -598,7 +598,7 @@ async def complete(model_arg, _messages, **_kwargs):


@pytest.mark.asyncio
async def test_invalid_write_file_arguments_request_three_protocol_repairs() -> None:
async def test_invalid_write_file_arguments_request_ten_protocol_repairs() -> None:
tenant_id = uuid.uuid4()
model = _model(tenant_id)
agent = _agent(tenant_id)
Expand Down
35 changes: 17 additions & 18 deletions backend/tests/test_agent_runtime_node_executor.py
Original file line number Diff line number Diff line change
Expand Up @@ -1351,15 +1351,16 @@ async def test_empty_output_is_repaired_once_then_fails_explicitly() -> None:

@pytest.mark.asyncio
@pytest.mark.parametrize(
("repair_code", "instruction"),
("repair_code", "instruction", "repair_limit"),
[
("invalid_finish", "Retry finish with valid content."),
("invalid_tool_call", "Retry with valid JSON tool arguments."),
("invalid_finish", "Retry finish with valid content.", 1),
("invalid_tool_call", "Retry with valid JSON tool arguments.", 10),
],
)
async def test_repeated_model_tool_protocol_repair_code_fails_explicitly(
repair_code: str,
instruction: str,
repair_limit: int,
) -> None:
run_id = uuid.uuid4()
repair = ModelStepResult(
Expand All @@ -1368,7 +1369,7 @@ async def test_repeated_model_tool_protocol_repair_code_fails_explicitly(
repair_instruction=instruction,
repair_code=repair_code,
)
model = ModelService(repair, repair)
model = ModelService(*([repair] * (repair_limit + 1)))
executor = _executor(model)

result = await _invoke(run_id, executor, model_turn_limit=50)
Expand All @@ -1377,13 +1378,13 @@ async def test_repeated_model_tool_protocol_repair_code_fails_explicitly(
assert lifecycle["status"] == "failed"
assert lifecycle["reason"] == "model_tool_protocol_violation"
assert lifecycle["error"]["code"] == "model_tool_protocol_violation"
assert lifecycle["model_protocol_repairs"] == {repair_code: 1}
assert lifecycle["model_step_count"] == 2
assert model.calls == 2
assert lifecycle["model_protocol_repairs"] == {repair_code: repair_limit}
assert lifecycle["model_step_count"] == repair_limit + 1
assert model.calls == repair_limit + 1


@pytest.mark.asyncio
async def test_write_file_protocol_repair_uses_three_attempts_then_guides_user() -> None:
async def test_write_file_protocol_repair_uses_ten_attempts_then_guides_user() -> None:
run_id = uuid.uuid4()
repair = ModelStepResult(
intent="text",
Expand All @@ -1392,7 +1393,7 @@ async def test_write_file_protocol_repair_uses_three_attempts_then_guides_user()
repair_code="invalid_tool_call",
repair_tool_name="write_file",
)
model = ModelService(repair, repair, repair, repair)
model = ModelService(*([repair] * 11))
executor = _executor(model)

result = await _invoke(run_id, executor, model_turn_limit=50)
Expand All @@ -1408,14 +1409,14 @@ async def test_write_file_protocol_repair_uses_three_attempts_then_guides_user()
),
}
assert lifecycle["model_protocol_repairs"] == {
"invalid_tool_call:write_file": 3,
"invalid_tool_call:write_file": 10,
}
assert lifecycle["model_step_count"] == 4
assert model.calls == 4
assert lifecycle["model_step_count"] == 11
assert model.calls == 11


@pytest.mark.asyncio
async def test_write_file_protocol_can_recover_on_the_third_repair() -> None:
async def test_write_file_protocol_can_recover_on_the_tenth_repair() -> None:
run_id = uuid.uuid4()
repair = ModelStepResult(
intent="text",
Expand All @@ -1424,9 +1425,7 @@ async def test_write_file_protocol_can_recover_on_the_third_repair() -> None:
repair_tool_name="write_file",
)
model = ModelService(
repair,
repair,
repair,
*([repair] * 10),
ModelStepResult(intent="finish", finish_content="Recovered"),
)
executor = _executor(model)
Expand All @@ -1435,9 +1434,9 @@ async def test_write_file_protocol_can_recover_on_the_third_repair() -> None:

assert result["lifecycle"]["status"] == "completed"
assert result["lifecycle"]["model_protocol_repairs"] == {
"invalid_tool_call:write_file": 3,
"invalid_tool_call:write_file": 10,
}
assert model.calls == 4
assert model.calls == 11


@pytest.mark.asyncio
Expand Down
4 changes: 2 additions & 2 deletions backend/tests/test_agent_runtime_tool_repair_budget.py
Original file line number Diff line number Diff line change
Expand Up @@ -50,7 +50,7 @@ def test_tenth_consecutive_fingerprint_pauses_without_off_by_one() -> None:
assert _episode(state)["total_failures"] == 10


def test_twentieth_tool_failure_pauses_even_when_fingerprint_changes() -> None:
def test_tenth_tool_failure_pauses_even_when_fingerprint_changes() -> None:
state: dict = {}
transition = None
for model_step in range(1, TOOL_EPISODE_FAILURE_LIMIT + 1):
Expand All @@ -63,7 +63,7 @@ def test_twentieth_tool_failure_pauses_even_when_fingerprint_changes() -> None:

assert transition is not None
assert transition.pause_reason == "tool_repair_episode_limit_reached"
assert _episode(state)["total_failures"] == 20
assert _episode(state)["total_failures"] == 10
assert _episode(state)["same_fingerprint_failures"] == 1


Expand Down
4 changes: 2 additions & 2 deletions backend/tests/test_agent_runtime_tool_step_service.py
Original file line number Diff line number Diff line change
Expand Up @@ -2977,7 +2977,7 @@ async def test_retryable_read_exhaustion_returns_one_non_retryable_result(
"call-read-exhausted",
"read_file",
)
execution.attempt_count = 3
execution.attempt_count = 10

async def reserve(db, **kwargs):
del db
Expand Down Expand Up @@ -3017,7 +3017,7 @@ async def mark_failed(db, **kwargs):
assert "Do not repeat the identical tool call unchanged" in result.messages[0][
"content"
]
assert execution.result_metadata["runtime_attempt_count"] == 3
assert execution.result_metadata["runtime_attempt_count"] == 10
assert execution.result_metadata["runtime_retry_exhausted"] is True
assert execution.result_metadata["last_error_code"] == "temporary_read_failure"

Expand Down
17 changes: 14 additions & 3 deletions backend/tests/test_chat_session_runtime_state.py
Original file line number Diff line number Diff line change
Expand Up @@ -239,15 +239,26 @@ async def test_runtime_state_exposes_unknown_write_and_blocks_plain_resume() ->


@pytest.mark.asyncio
async def test_runtime_state_exposes_unknown_image_generation_for_user_confirmation() -> None:
@pytest.mark.parametrize(
("tool_name", "contract_version"),
[
("generate_image_openai", None),
("tenant_search", "registered:tenant_search:0123456789abcdef"),
],
)
async def test_runtime_state_exposes_reconcilable_unknown_tool_for_user_confirmation(
tool_name: str,
contract_version: str | None,
) -> None:
agent, user, session, run = _records()
reader = SimpleNamespace(get_run_state=AsyncMock(return_value=_view(run)))
execution = AgentToolExecution(
id=uuid.uuid4(),
tenant_id=run.tenant_id,
run_id=run.id,
tool_call_id="call-image-1",
tool_name="generate_image_openai",
tool_name=tool_name,
contract_version=contract_version,
assistant_message_id="assistant-1",
arguments_hash="hash",
sanitized_arguments={},
Expand Down Expand Up @@ -287,7 +298,7 @@ async def test_runtime_state_exposes_unknown_image_generation_for_user_confirmat

assert response.active_run is not None
assert response.active_run.can_resume is False
assert response.active_run.pending_tool_reconciliations[0].tool_name == "generate_image_openai"
assert response.active_run.pending_tool_reconciliations[0].tool_name == tool_name
assert response.active_run.pending_tool_reconciliations[0].can_reconcile is True


Expand Down
10 changes: 5 additions & 5 deletions backend/tests/test_finish_protocol.py
Original file line number Diff line number Diff line change
Expand Up @@ -811,7 +811,7 @@ async def test_repeated_invalid_tool_json_is_bounded_by_protocol_code(monkeypatc
}
],
)
fake_client = FakeStreamClient([invalid, invalid])
fake_client = FakeStreamClient([invalid] * 11)
monkeypatch.setattr(caller, "_get_agent_config", lambda _agent_id: _async_return((50, None)))
monkeypatch.setattr(caller, "_get_user_name", lambda _user_id: _async_return("Ray"))
monkeypatch.setattr(
Expand Down Expand Up @@ -841,12 +841,12 @@ async def test_repeated_invalid_tool_json_is_bounded_by_protocol_code(monkeypatc
)

assert result.startswith("[Error] invalid_tool_call_protocol_violation:")
assert len(fake_client.messages_seen) == 2
assert len(fake_client.messages_seen) == 11
assert fake_client.closed is True


@pytest.mark.asyncio
async def test_invalid_write_file_json_gets_three_bounded_repairs(monkeypatch):
async def test_invalid_write_file_json_gets_ten_bounded_repairs(monkeypatch):
from app.services.llm import caller
from app.services.llm.client import LLMResponse

Expand All @@ -863,7 +863,7 @@ async def test_invalid_write_file_json_gets_three_bounded_repairs(monkeypatch):
}
],
)
fake_client = FakeStreamClient([invalid, invalid, invalid, invalid])
fake_client = FakeStreamClient([invalid] * 11)
monkeypatch.setattr(caller, "_get_agent_config", lambda _agent_id: _async_return((50, None)))
monkeypatch.setattr(caller, "_get_user_name", lambda _user_id: _async_return("Ray"))
monkeypatch.setattr(
Expand Down Expand Up @@ -897,7 +897,7 @@ async def test_invalid_write_file_json_gets_three_bounded_repairs(monkeypatch):
"本次文件生成未完成:write_file 工具参数无效或被截断,连续重试后仍无法执行。"
"请回复「重新生成」,我会基于当前对话重新尝试。"
)
assert len(fake_client.messages_seen) == 4
assert len(fake_client.messages_seen) == 11
assert fake_client.closed is True


Expand Down
21 changes: 16 additions & 5 deletions backend/tests/test_tool_execution.py
Original file line number Diff line number Diff line change
Expand Up @@ -187,10 +187,16 @@ def _sql(statement) -> str:
],
)
@pytest.mark.parametrize(
("tool_name", "effect", "retry_policy"),
("tool_name", "effect", "retry_policy", "contract_version"),
[
("write_file", "write", "conditional"),
("generate_image_openai", "external_write", "never"),
("write_file", "write", "conditional", None),
("generate_image_openai", "external_write", "never", None),
(
"tenant_search",
"external_write",
"never",
"registered:tenant_search:0123456789abcdef",
),
],
)
async def test_user_reconcilable_unknown_receipt_can_be_settled(
Expand All @@ -199,6 +205,7 @@ async def test_user_reconcilable_unknown_receipt_can_be_settled(
tool_name: str,
effect: str,
retry_policy: str,
contract_version: str | None,
) -> None:
tenant_id = uuid.uuid4()
run_id = uuid.uuid4()
Expand All @@ -211,6 +218,7 @@ async def test_user_reconcilable_unknown_receipt_can_be_settled(
retry_policy=retry_policy,
)
execution.tool_name = tool_name
execution.contract_version = contract_version
execution.completed_at = _NOW
db = _FakeSession(execution)

Expand Down Expand Up @@ -249,7 +257,7 @@ async def test_unknown_reconciliation_rejects_unsupported_tool() -> None:

with pytest.raises(
tool_execution.ToolExecutionError,
match="only supported for conditional write_file or image-generation",
match="not supported for this Tool receipt",
):
await tool_execution.reconcile_unknown_tool_execution(
db, # type: ignore[arg-type]
Expand Down Expand Up @@ -881,7 +889,10 @@ async def test_expired_final_safe_read_attempt_closes_without_provider_replay():
assert reservation.prior_failure is not None
assert reservation.prior_failure.error_code == "tool_retry_exhausted"
assert execution.status == "failed"
assert execution.result_metadata["runtime_attempt_count"] == 3
assert (
execution.result_metadata["runtime_attempt_count"]
== tool_execution.SAFE_READ_MAX_ATTEMPTS
)
assert execution.result_metadata["runtime_retry_exhausted"] is True
assert db.flush_count == 1

Expand Down
2 changes: 1 addition & 1 deletion specs/002-tool-runtime-contract/checklists/requirements.md
Original file line number Diff line number Diff line change
Expand Up @@ -33,4 +33,4 @@

- 第一次校验即通过,无 `[NEEDS CLARIFICATION]` 项。
- `Tool Call`、`Run`、`Receipt`、`checkpoint` 等词是本产品领域对象,不是具体实现方案;具体数据结构、文件和迁移步骤将在 Plan 阶段定义。
- Spec 已覆盖用户确认的 10/20 repair budget、模型可见错误反馈、unknown write 禁止自动重放和旧 checkpoint 兼容边界。
- Spec 已覆盖用户确认的 Tool repair/retry 上限统一为 10、模型可见错误反馈、unknown write 禁止自动重放和旧 checkpoint 兼容边界;计数结构统一重构已明确延期
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,8 @@
## Tool Repair Episode

- `same_fingerprint_failures` reaches 10: pause immediately after recording the 10th failure; do not invoke model step 11 for that loop.
- `total_failures` reaches 20 for the same Tool episode: pause immediately; do not invoke the next model step.
- `total_failures` reaches 10 for the same Tool episode: pause immediately; do not invoke the next model step.
- Generic Tool protocol repair, `write_file` protocol repair, and safe-read replay retain their current independent counters but each uses a limit of 10; counter unification is deferred.
- Changing fingerprint resets only the consecutive counter.
- Success of the same Tool, new Run, or explicit user correction resets the Tool episode.
- Success of another Tool does not reset it.
Expand Down
Loading