From fb9558439ca4ff782ffbe21fe5d12834fb67cad7 Mon Sep 17 00:00:00 2001 From: pewdiepie-archdaemon Date: Fri, 18 Sep 2026 13:29:54 +0000 Subject: [PATCH] fix(agent): preserve tool evidence through final synthesis --- src/clean_agent_preview.py | 157 +++++++++++++++++- src/tool_parsing.py | 7 +- tests/test_clean_agent_preview.py | 36 +++- ..._example_not_executed_for_native_models.py | 17 ++ 4 files changed, 202 insertions(+), 15 deletions(-) diff --git a/src/clean_agent_preview.py b/src/clean_agent_preview.py index f10079921..03bff0118 100644 --- a/src/clean_agent_preview.py +++ b/src/clean_agent_preview.py @@ -30,6 +30,7 @@ from src.tool_schemas import ( normalized_native_function_argument_error, ) from src.tool_types import ToolBlock +from src.tool_parsing import parse_tool_blocks, strip_tool_blocks from src.turn_contract import ( FAMILY_TOOLS, broad_web_briefing_request, required_read_operation_for_request, targets_bound_editor_request, inline_text_transformation, @@ -684,7 +685,21 @@ def documents_terminal_response(raw, *, user_text='', max_items=8): def shell_listing_terminal_response(raw, *, user_text=''): """Return successful read-only listing evidence when prose omits the rows.""" - if not re.search(r'\b(?:list|names?)\b', str(user_text or ''), re.I): + # Bare words such as "list every move" or "list the findings" describe + # the shape of the eventual answer; they do not ask for shell stdout. The + # old broad matcher made an intermediate ``ls`` (for example, after video + # frame extraction) terminate the whole agent turn as "Workspace items". + # Only let the shell own rendering when the user explicitly requested a + # filesystem/directory listing. + if not re.search( + r'\b(?:list|show|display|name)\b[^\n]{0,48}' + r'\b(?:files?|folders?|director(?:y|ies)|workspace items?|paths?|filenames?)\b' + r'|\b(?:files?|folders?|director(?:y|ies)|workspace items?|paths?|filenames?)\b' + r'[^\n]{0,48}\b(?:list|names?)\b' + r'|\blist\b[^\n]{0,32}\b(?:whats|what\'s|what is)\s+in\s+there\b', + str(user_text or ''), + re.I, + ): return '' payload = raw try: @@ -718,6 +733,19 @@ def shell_output_terminal_response(raw, *, maximum=4000): return text[:maximum] + ('\n…' if len(text) > maximum else '') +def direct_shell_output_request(user_text): + """Whether raw stdout itself is the deliverable requested by the user.""" + text = str(user_text or '') + return bool(re.search( + r'\b(?:run|execute)\b[^\n]{0,32}\b(?:this\s+)?(?:command|script)\b' + r'|\b(?:bash|shell|terminal|stdout|command output)\b' + r'|\b(?:print|show|tell me|what(?:\'s| is))\b[^\n]{0,40}' + r'\b(?:hostname|working directory|current directory|workspace path|pwd)\b', + text, + re.I, + )) + + def ui_panel_terminal_response(raw, *, args=None): """Render only UI state confirmed by a successful ui_control result.""" command = dict(args or {}) @@ -4116,6 +4144,8 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac official_source_retry_attempted = False note_search_recovery_attempted = False replace_streamed_draft_on_finish = False + final_synthesis_reserved = False + emergency_completion_round = False # Stream model text immediately. A canonical final event reconciles any # draft that completion/research checks subsequently replace. finalize_search_answer = broad_current_web_request(direct_user_text) or requested_web_source_links(direct_user_text) @@ -4167,9 +4197,46 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac ), limits=preview_http_limits(), ) as client: - for round_number in range(1, round_limit + 1): + # One extra iteration is available only when a provider emits raw + # tool markup during the normal final no-tools round. Ordinary + # turns still obey ``round_limit`` exactly. + for round_number in range(1, round_limit + 2): + if round_number > round_limit and not emergency_completion_round: + break rounds_used = round_number yield event({'type': 'agent_step', 'round': round_number}) + # Preserve the final model round for an actual user-facing + # answer once tools have returned evidence. Previously the + # model could spend the last round emitting another tool call + # (or provider-native tool markup that was rendered as prose), + # leaving no opportunity to synthesize the result. + reserve_final_synthesis = bool( + round_number >= round_limit + and executions + and (not required_artifacts or successful_artifact_write) + ) + if reserve_final_synthesis and not final_synthesis_reserved: + final_synthesis_reserved = True + final_instruction = ( + 'Final completion round: no more tools are available. Finish from the ' + 'evidence already returned and answer the user directly and completely now. ' + 'Do not emit tool-call markup, describe another planned action, or merely ' + 'repeat raw tool output. State any remaining uncertainty explicitly.' + ) + if history and history[-1].get('_harness_control'): + history[-1]['content'] = ( + str(history[-1].get('content') or '') + ' ' + final_instruction + ) + else: + history.append({ + 'role': 'user', + '_harness_control': True, + 'content': final_instruction, + }) + yield event({ + 'type': 'completion_recovery', + 'reason': 'reserved_final_synthesis_round', + }) # Enforce a known research prerequisite before asking the model # for another response, not after streaming a premature answer. if ( @@ -4218,7 +4285,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac request_messages = provider_request_messages( prune_multimodal_images(history, max_images=3) ) - if prior_summary_answer or force_no_tools_next_round: + if prior_summary_answer or force_no_tools_next_round or reserve_final_synthesis: round_offered = [] force_no_tools_next_round = False else: @@ -4380,6 +4447,55 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac if content and not prior_summary_answer: yield event({'delta': content}) proposed = [pending[i] for i in sorted(pending)] + unexecutable_dsml_completion = False + if not proposed and 'DSML' in content: + offered_by_canonical = { + canonical(schema['function']['name']): schema['function']['name'] + for schema in round_offered + } + parsed_dsml_blocks = parse_tool_blocks( + content, + skip_fenced=True, + additional_tool_names=offered_by_canonical.values(), + additional_tool_schemas=round_offered, + ) + recovered = [] + for index, block in enumerate(parsed_dsml_blocks): + actual_name = offered_by_canonical.get(canonical(block.tool_type)) + if not actual_name: + continue + try: + recovered_args = json.loads(block.content or '{}') + except (TypeError, ValueError, json.JSONDecodeError): + continue + if not isinstance(recovered_args, dict): + continue + recovered.append({ + 'id': f'call_dsml_{round_number}_{index}', + 'type': 'function', + 'function': { + 'name': actual_name, + 'arguments': json.dumps(recovered_args, ensure_ascii=False), + }, + }) + if recovered: + proposed = recovered + if parsed_dsml_blocks: + content = strip_tool_blocks( + content, + skip_fenced=True, + additional_tool_names=offered_by_canonical.values(), + ).strip() + replace_streamed_draft_on_finish = True + if recovered: + yield event({ + 'type': 'tool_markup_recovery', + 'format': 'deepseek_dsml', + 'round': round_number, + 'calls': len(recovered), + }) + else: + unexecutable_dsml_completion = True proposed = serialize_required_email_attachment_chain( proposed, contract_required_tools, executions, ) @@ -4392,6 +4508,33 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac message['tool_calls'] = protocol_safe_tool_calls(proposed) history.append(message) if not proposed: + if ( + unexecutable_dsml_completion + and ( + round_number < round_limit + or (round_number == round_limit and not emergency_completion_round) + ) + ): + answer_recovery_attempts += 1 + force_no_tools_next_round = True + if round_number == round_limit: + emergency_completion_round = True + history.pop() + history.append({ + 'role': 'user', + '_harness_control': True, + 'content': ( + 'Your draft emitted tool-call markup during the no-tools completion ' + 'phase. That call was not executed. Do not emit DSML, XML, JSON tool ' + 'calls, or another action plan. Answer the original request directly ' + 'now from the evidence already present, and state uncertainty plainly.' + ), + }) + yield event({ + 'type': 'completion_recovery', + 'reason': 'tool_markup_during_final_synthesis', + }) + continue if prior_summary_answer: content = prior_summary_answer history[-1]['content'] = content @@ -5610,7 +5753,9 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac ): shell_terminal_response = shell_listing_terminal_response( output, user_text=latest_user, - ) or shell_output_terminal_response(output) + ) + if not shell_terminal_response and direct_shell_output_request(latest_user): + shell_terminal_response = shell_output_terminal_response(output) artifact_pending = bool( required_artifacts and not successful_artifact_write ) @@ -5817,9 +5962,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac break if terminal_budget_violation: if ( - getattr(turn_contract, 'routing_experiment', '') == MODEL_CHOICE_MODE - and not native_workspace_enabled - and not budget_completion_attempted + not budget_completion_attempted and round_number < round_limit ): budget_completion_attempted = True diff --git a/src/tool_parsing.py b/src/tool_parsing.py index 6cd237c20..985100ce9 100644 --- a/src/tool_parsing.py +++ b/src/tool_parsing.py @@ -268,8 +268,11 @@ def _normalize_dsml(text: str) -> str: if "DSML" not in text: return text t = text - t = re.sub(rf"<\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*tool_calls\s*>", "", t, flags=re.IGNORECASE) - t = re.sub(rf"<\s*/\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*tool_calls\s*>", "", t, flags=re.IGNORECASE) + # Hosted DeepSeek variants use both ``tool_calls`` and the shorter + # ``calls`` wrapper. Treat them identically; otherwise the inner invoke is + # parsed but the outer DSML tags leak into the user-visible response. + t = re.sub(rf"<\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*(?:tool_calls|calls)\s*>", "", t, flags=re.IGNORECASE) + t = re.sub(rf"<\s*/\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*(?:tool_calls|calls)\s*>", "", t, flags=re.IGNORECASE) t = re.sub(rf"<\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*invoke\s+name=", "", "", t, flags=re.IGNORECASE) # parameter open tag — drop any extra attrs (e.g. string="true"). diff --git a/tests/test_clean_agent_preview.py b/tests/test_clean_agent_preview.py index 3be6d49fa..4937c885f 100644 --- a/tests/test_clean_agent_preview.py +++ b/tests/test_clean_agent_preview.py @@ -5,7 +5,7 @@ import jsonschema import pytest import re -from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, execution_targets_required_artifact, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, bounded_web_evidence_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, web_fetch_observation_is_boilerplate, broad_current_web_request, record_tool_execution, align_structured_tool_history, provider_request_messages, provider_compatible_tool_choice_request, offered_tool_alias, dependent_write_prerequisite_error, bounded_research_tool_policy, retrieved_source_urls, serialize_required_email_attachment_chain +from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, execution_targets_required_artifact, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, direct_shell_output_request, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, bounded_web_evidence_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, web_fetch_observation_is_boilerplate, broad_current_web_request, record_tool_execution, align_structured_tool_history, provider_request_messages, provider_compatible_tool_choice_request, offered_tool_alias, dependent_write_prerequisite_error, bounded_research_tool_policy, retrieved_source_urls, serialize_required_email_attachment_chain from src.tool_capabilities import capabilities_for_tool @@ -1468,6 +1468,21 @@ def test_shell_listing_renderer_uses_successful_stdout_rows(): assert rendered == "Workspace items (3):\n- alpha\n- beta\n- gamma" +def test_shell_listing_renderer_does_not_mistake_requested_answer_list_for_files(): + assert shell_listing_terminal_response( + "f_001.png\nf_002.png\n2", + user_text="Inspect the video and list every chess move with timestamps.", + ) == "" + + +def test_raw_shell_stdout_only_owns_explicit_shell_requests(): + assert direct_shell_output_request("Run this command and return its stdout: uname -a") + assert direct_shell_output_request("What is the current working directory?") + assert not direct_shell_output_request( + "Inspect the poster, calculate the package price, and explain ambiguities." + ) + + def test_workspace_path_followup_reuses_prior_pwd_evidence(): history = [ {'role': 'tool', 'content': '/workspace'}, @@ -4210,7 +4225,7 @@ async def test_native_stream_explains_how_to_recover_from_timestamp_free_export( @pytest.mark.asyncio -async def test_native_stream_terminates_on_first_post_budget_tool_call(monkeypatch): +async def test_native_stream_synthesizes_after_first_post_budget_tool_call(monkeypatch): import src.clean_agent_preview as module responses = iter([ {"choices": [{"delta": {"tool_calls": [{ @@ -4230,6 +4245,7 @@ async def test_native_stream_terminates_on_first_post_budget_tool_call(monkeypat }), }, }]}}]}, + {"choices": [{"delta": {"content": "The visual evidence shows a complete result."}}]}, ]) class Response: @@ -4282,8 +4298,13 @@ async def test_native_stream_terminates_on_first_post_budget_tool_call(monkeypat ] assert len(budget_errors) == 1 final = [event for event in events if event.get("type") == "final_response"] - assert len(final) == 1 - assert "budget was exhausted" in final[0]["content"] + assert not final + assert any( + event.get("type") == "completion_recovery" + and event.get("reason") == "tool_budget_final_synthesis" + for event in events + ) + assert any("The visual evidence shows a complete result." in chunk for chunk in raw) @pytest.mark.asyncio @@ -5808,5 +5829,8 @@ async def test_parallel_tool_results_precede_visual_evidence(monkeypatch): messages = requests[1]["messages"] assistant_index = max(i for i, message in enumerate(messages) if message["role"] == "assistant") - assert [message["role"] for message in messages[assistant_index + 1:]] == ["tool", "tool", "user"] - assert messages[-1]["content"][1]["type"] == "image_url" + assert [message["role"] for message in messages[assistant_index + 1:]] == [ + "tool", "tool", "user", "user", + ] + assert "Final completion round" in messages[-1]["content"] + assert messages[-2]["content"][1]["type"] == "image_url" diff --git a/tests/test_fenced_example_not_executed_for_native_models.py b/tests/test_fenced_example_not_executed_for_native_models.py index 198cedb68..67cb78843 100644 --- a/tests/test_fenced_example_not_executed_for_native_models.py +++ b/tests/test_fenced_example_not_executed_for_native_models.py @@ -449,6 +449,23 @@ def test_skip_fenced_still_recovers_dsml_markup(): assert "latest python release" in blocks[0].content +def test_short_dsml_calls_wrapper_is_parsed_and_fully_stripped(): + dsml = ( + "Evidence gathered.\n" + "<||DSML|| calls>" + '<||DSML|| invoke name="extract_text">' + '<||DSML|| parameter name="path" string="true">/workspace/frame.png' + '' + "" + "" + ) + blocks = parse_tool_blocks(dsml, skip_fenced=True, additional_tool_names=["extract_text"]) + assert len(blocks) == 1 + assert blocks[0].tool_type == "extract_text" + assert json.loads(blocks[0].content) == {"path": "/workspace/frame.png"} + assert strip_tool_blocks(dsml, skip_fenced=True) == "Evidence gathered." + + def test_skip_fenced_ignores_only_the_fenced_pattern(): text = "```bash\nnpm run plan:articles\n```" assert parse_tool_blocks(text, skip_fenced=True) == []