diff --git a/src/clean_agent_preview.py b/src/clean_agent_preview.py
index f10079921..03bff0118 100644
--- a/src/clean_agent_preview.py
+++ b/src/clean_agent_preview.py
@@ -30,6 +30,7 @@ from src.tool_schemas import (
normalized_native_function_argument_error,
)
from src.tool_types import ToolBlock
+from src.tool_parsing import parse_tool_blocks, strip_tool_blocks
from src.turn_contract import (
FAMILY_TOOLS, broad_web_briefing_request, required_read_operation_for_request,
targets_bound_editor_request, inline_text_transformation,
@@ -684,7 +685,21 @@ def documents_terminal_response(raw, *, user_text='', max_items=8):
def shell_listing_terminal_response(raw, *, user_text=''):
"""Return successful read-only listing evidence when prose omits the rows."""
- if not re.search(r'\b(?:list|names?)\b', str(user_text or ''), re.I):
+ # Bare words such as "list every move" or "list the findings" describe
+ # the shape of the eventual answer; they do not ask for shell stdout. The
+ # old broad matcher made an intermediate ``ls`` (for example, after video
+ # frame extraction) terminate the whole agent turn as "Workspace items".
+ # Only let the shell own rendering when the user explicitly requested a
+ # filesystem/directory listing.
+ if not re.search(
+ r'\b(?:list|show|display|name)\b[^\n]{0,48}'
+ r'\b(?:files?|folders?|director(?:y|ies)|workspace items?|paths?|filenames?)\b'
+ r'|\b(?:files?|folders?|director(?:y|ies)|workspace items?|paths?|filenames?)\b'
+ r'[^\n]{0,48}\b(?:list|names?)\b'
+ r'|\blist\b[^\n]{0,32}\b(?:whats|what\'s|what is)\s+in\s+there\b',
+ str(user_text or ''),
+ re.I,
+ ):
return ''
payload = raw
try:
@@ -718,6 +733,19 @@ def shell_output_terminal_response(raw, *, maximum=4000):
return text[:maximum] + ('\n…' if len(text) > maximum else '')
+def direct_shell_output_request(user_text):
+ """Whether raw stdout itself is the deliverable requested by the user."""
+ text = str(user_text or '')
+ return bool(re.search(
+ r'\b(?:run|execute)\b[^\n]{0,32}\b(?:this\s+)?(?:command|script)\b'
+ r'|\b(?:bash|shell|terminal|stdout|command output)\b'
+ r'|\b(?:print|show|tell me|what(?:\'s| is))\b[^\n]{0,40}'
+ r'\b(?:hostname|working directory|current directory|workspace path|pwd)\b',
+ text,
+ re.I,
+ ))
+
+
def ui_panel_terminal_response(raw, *, args=None):
"""Render only UI state confirmed by a successful ui_control result."""
command = dict(args or {})
@@ -4116,6 +4144,8 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
official_source_retry_attempted = False
note_search_recovery_attempted = False
replace_streamed_draft_on_finish = False
+ final_synthesis_reserved = False
+ emergency_completion_round = False
# Stream model text immediately. A canonical final event reconciles any
# draft that completion/research checks subsequently replace.
finalize_search_answer = broad_current_web_request(direct_user_text) or requested_web_source_links(direct_user_text)
@@ -4167,9 +4197,46 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
),
limits=preview_http_limits(),
) as client:
- for round_number in range(1, round_limit + 1):
+ # One extra iteration is available only when a provider emits raw
+ # tool markup during the normal final no-tools round. Ordinary
+ # turns still obey ``round_limit`` exactly.
+ for round_number in range(1, round_limit + 2):
+ if round_number > round_limit and not emergency_completion_round:
+ break
rounds_used = round_number
yield event({'type': 'agent_step', 'round': round_number})
+ # Preserve the final model round for an actual user-facing
+ # answer once tools have returned evidence. Previously the
+ # model could spend the last round emitting another tool call
+ # (or provider-native tool markup that was rendered as prose),
+ # leaving no opportunity to synthesize the result.
+ reserve_final_synthesis = bool(
+ round_number >= round_limit
+ and executions
+ and (not required_artifacts or successful_artifact_write)
+ )
+ if reserve_final_synthesis and not final_synthesis_reserved:
+ final_synthesis_reserved = True
+ final_instruction = (
+ 'Final completion round: no more tools are available. Finish from the '
+ 'evidence already returned and answer the user directly and completely now. '
+ 'Do not emit tool-call markup, describe another planned action, or merely '
+ 'repeat raw tool output. State any remaining uncertainty explicitly.'
+ )
+ if history and history[-1].get('_harness_control'):
+ history[-1]['content'] = (
+ str(history[-1].get('content') or '') + ' ' + final_instruction
+ )
+ else:
+ history.append({
+ 'role': 'user',
+ '_harness_control': True,
+ 'content': final_instruction,
+ })
+ yield event({
+ 'type': 'completion_recovery',
+ 'reason': 'reserved_final_synthesis_round',
+ })
# Enforce a known research prerequisite before asking the model
# for another response, not after streaming a premature answer.
if (
@@ -4218,7 +4285,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
request_messages = provider_request_messages(
prune_multimodal_images(history, max_images=3)
)
- if prior_summary_answer or force_no_tools_next_round:
+ if prior_summary_answer or force_no_tools_next_round or reserve_final_synthesis:
round_offered = []
force_no_tools_next_round = False
else:
@@ -4380,6 +4447,55 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
if content and not prior_summary_answer:
yield event({'delta': content})
proposed = [pending[i] for i in sorted(pending)]
+ unexecutable_dsml_completion = False
+ if not proposed and 'DSML' in content:
+ offered_by_canonical = {
+ canonical(schema['function']['name']): schema['function']['name']
+ for schema in round_offered
+ }
+ parsed_dsml_blocks = parse_tool_blocks(
+ content,
+ skip_fenced=True,
+ additional_tool_names=offered_by_canonical.values(),
+ additional_tool_schemas=round_offered,
+ )
+ recovered = []
+ for index, block in enumerate(parsed_dsml_blocks):
+ actual_name = offered_by_canonical.get(canonical(block.tool_type))
+ if not actual_name:
+ continue
+ try:
+ recovered_args = json.loads(block.content or '{}')
+ except (TypeError, ValueError, json.JSONDecodeError):
+ continue
+ if not isinstance(recovered_args, dict):
+ continue
+ recovered.append({
+ 'id': f'call_dsml_{round_number}_{index}',
+ 'type': 'function',
+ 'function': {
+ 'name': actual_name,
+ 'arguments': json.dumps(recovered_args, ensure_ascii=False),
+ },
+ })
+ if recovered:
+ proposed = recovered
+ if parsed_dsml_blocks:
+ content = strip_tool_blocks(
+ content,
+ skip_fenced=True,
+ additional_tool_names=offered_by_canonical.values(),
+ ).strip()
+ replace_streamed_draft_on_finish = True
+ if recovered:
+ yield event({
+ 'type': 'tool_markup_recovery',
+ 'format': 'deepseek_dsml',
+ 'round': round_number,
+ 'calls': len(recovered),
+ })
+ else:
+ unexecutable_dsml_completion = True
proposed = serialize_required_email_attachment_chain(
proposed, contract_required_tools, executions,
)
@@ -4392,6 +4508,33 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
message['tool_calls'] = protocol_safe_tool_calls(proposed)
history.append(message)
if not proposed:
+ if (
+ unexecutable_dsml_completion
+ and (
+ round_number < round_limit
+ or (round_number == round_limit and not emergency_completion_round)
+ )
+ ):
+ answer_recovery_attempts += 1
+ force_no_tools_next_round = True
+ if round_number == round_limit:
+ emergency_completion_round = True
+ history.pop()
+ history.append({
+ 'role': 'user',
+ '_harness_control': True,
+ 'content': (
+ 'Your draft emitted tool-call markup during the no-tools completion '
+ 'phase. That call was not executed. Do not emit DSML, XML, JSON tool '
+ 'calls, or another action plan. Answer the original request directly '
+ 'now from the evidence already present, and state uncertainty plainly.'
+ ),
+ })
+ yield event({
+ 'type': 'completion_recovery',
+ 'reason': 'tool_markup_during_final_synthesis',
+ })
+ continue
if prior_summary_answer:
content = prior_summary_answer
history[-1]['content'] = content
@@ -5610,7 +5753,9 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
):
shell_terminal_response = shell_listing_terminal_response(
output, user_text=latest_user,
- ) or shell_output_terminal_response(output)
+ )
+ if not shell_terminal_response and direct_shell_output_request(latest_user):
+ shell_terminal_response = shell_output_terminal_response(output)
artifact_pending = bool(
required_artifacts and not successful_artifact_write
)
@@ -5817,9 +5962,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac
break
if terminal_budget_violation:
if (
- getattr(turn_contract, 'routing_experiment', '') == MODEL_CHOICE_MODE
- and not native_workspace_enabled
- and not budget_completion_attempted
+ not budget_completion_attempted
and round_number < round_limit
):
budget_completion_attempted = True
diff --git a/src/tool_parsing.py b/src/tool_parsing.py
index 6cd237c20..985100ce9 100644
--- a/src/tool_parsing.py
+++ b/src/tool_parsing.py
@@ -268,8 +268,11 @@ def _normalize_dsml(text: str) -> str:
if "DSML" not in text:
return text
t = text
- t = re.sub(rf"<\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*tool_calls\s*>", "", t, flags=re.IGNORECASE)
- t = re.sub(rf"<\s*/\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*tool_calls\s*>", "", t, flags=re.IGNORECASE)
+ # Hosted DeepSeek variants use both ``tool_calls`` and the shorter
+ # ``calls`` wrapper. Treat them identically; otherwise the inner invoke is
+ # parsed but the outer DSML tags leak into the user-visible response.
+ t = re.sub(rf"<\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*(?:tool_calls|calls)\s*>", "", t, flags=re.IGNORECASE)
+ t = re.sub(rf"<\s*/\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*(?:tool_calls|calls)\s*>", "", t, flags=re.IGNORECASE)
t = re.sub(rf"<\s*{_DSML_PIPES}\s*DSML\s*{_DSML_PIPES}\s*invoke\s+name=", "", "", t, flags=re.IGNORECASE)
# parameter open tag — drop any extra attrs (e.g. string="true").
diff --git a/tests/test_clean_agent_preview.py b/tests/test_clean_agent_preview.py
index 3be6d49fa..4937c885f 100644
--- a/tests/test_clean_agent_preview.py
+++ b/tests/test_clean_agent_preview.py
@@ -5,7 +5,7 @@ import jsonschema
import pytest
import re
-from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, execution_targets_required_artifact, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, bounded_web_evidence_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, web_fetch_observation_is_boilerplate, broad_current_web_request, record_tool_execution, align_structured_tool_history, provider_request_messages, provider_compatible_tool_choice_request, offered_tool_alias, dependent_write_prerequisite_error, bounded_research_tool_policy, retrieved_source_urls, serialize_required_email_attachment_chain
+from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, execution_targets_required_artifact, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, direct_shell_output_request, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, bounded_web_evidence_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, web_fetch_observation_is_boilerplate, broad_current_web_request, record_tool_execution, align_structured_tool_history, provider_request_messages, provider_compatible_tool_choice_request, offered_tool_alias, dependent_write_prerequisite_error, bounded_research_tool_policy, retrieved_source_urls, serialize_required_email_attachment_chain
from src.tool_capabilities import capabilities_for_tool
@@ -1468,6 +1468,21 @@ def test_shell_listing_renderer_uses_successful_stdout_rows():
assert rendered == "Workspace items (3):\n- alpha\n- beta\n- gamma"
+def test_shell_listing_renderer_does_not_mistake_requested_answer_list_for_files():
+ assert shell_listing_terminal_response(
+ "f_001.png\nf_002.png\n2",
+ user_text="Inspect the video and list every chess move with timestamps.",
+ ) == ""
+
+
+def test_raw_shell_stdout_only_owns_explicit_shell_requests():
+ assert direct_shell_output_request("Run this command and return its stdout: uname -a")
+ assert direct_shell_output_request("What is the current working directory?")
+ assert not direct_shell_output_request(
+ "Inspect the poster, calculate the package price, and explain ambiguities."
+ )
+
+
def test_workspace_path_followup_reuses_prior_pwd_evidence():
history = [
{'role': 'tool', 'content': '/workspace'},
@@ -4210,7 +4225,7 @@ async def test_native_stream_explains_how_to_recover_from_timestamp_free_export(
@pytest.mark.asyncio
-async def test_native_stream_terminates_on_first_post_budget_tool_call(monkeypatch):
+async def test_native_stream_synthesizes_after_first_post_budget_tool_call(monkeypatch):
import src.clean_agent_preview as module
responses = iter([
{"choices": [{"delta": {"tool_calls": [{
@@ -4230,6 +4245,7 @@ async def test_native_stream_terminates_on_first_post_budget_tool_call(monkeypat
}),
},
}]}}]},
+ {"choices": [{"delta": {"content": "The visual evidence shows a complete result."}}]},
])
class Response:
@@ -4282,8 +4298,13 @@ async def test_native_stream_terminates_on_first_post_budget_tool_call(monkeypat
]
assert len(budget_errors) == 1
final = [event for event in events if event.get("type") == "final_response"]
- assert len(final) == 1
- assert "budget was exhausted" in final[0]["content"]
+ assert not final
+ assert any(
+ event.get("type") == "completion_recovery"
+ and event.get("reason") == "tool_budget_final_synthesis"
+ for event in events
+ )
+ assert any("The visual evidence shows a complete result." in chunk for chunk in raw)
@pytest.mark.asyncio
@@ -5808,5 +5829,8 @@ async def test_parallel_tool_results_precede_visual_evidence(monkeypatch):
messages = requests[1]["messages"]
assistant_index = max(i for i, message in enumerate(messages) if message["role"] == "assistant")
- assert [message["role"] for message in messages[assistant_index + 1:]] == ["tool", "tool", "user"]
- assert messages[-1]["content"][1]["type"] == "image_url"
+ assert [message["role"] for message in messages[assistant_index + 1:]] == [
+ "tool", "tool", "user", "user",
+ ]
+ assert "Final completion round" in messages[-1]["content"]
+ assert messages[-2]["content"][1]["type"] == "image_url"
diff --git a/tests/test_fenced_example_not_executed_for_native_models.py b/tests/test_fenced_example_not_executed_for_native_models.py
index 198cedb68..67cb78843 100644
--- a/tests/test_fenced_example_not_executed_for_native_models.py
+++ b/tests/test_fenced_example_not_executed_for_native_models.py
@@ -449,6 +449,23 @@ def test_skip_fenced_still_recovers_dsml_markup():
assert "latest python release" in blocks[0].content
+def test_short_dsml_calls_wrapper_is_parsed_and_fully_stripped():
+ dsml = (
+ "Evidence gathered.\n"
+ "<||DSML|| calls>"
+ '<||DSML|| invoke name="extract_text">'
+ '<||DSML|| parameter name="path" string="true">/workspace/frame.png'
+ '||DSML|| parameter>'
+ "||DSML|| invoke>"
+ "||DSML|| calls>"
+ )
+ blocks = parse_tool_blocks(dsml, skip_fenced=True, additional_tool_names=["extract_text"])
+ assert len(blocks) == 1
+ assert blocks[0].tool_type == "extract_text"
+ assert json.loads(blocks[0].content) == {"path": "/workspace/frame.png"}
+ assert strip_tool_blocks(dsml, skip_fenced=True) == "Evidence gathered."
+
+
def test_skip_fenced_ignores_only_the_fenced_pattern():
text = "```bash\nnpm run plan:articles\n```"
assert parse_tool_blocks(text, skip_fenced=True) == []