From dfd4d5a80e0037b87dac445e6fa3d25788137b75 Mon Sep 17 00:00:00 2001 From: pewdiepie-archdaemon Date: Thu, 17 Sep 2026 19:35:54 +0000 Subject: [PATCH] preserve evidence after bounded web search --- src/clean_agent_preview.py | 64 ++++++++++++++++++++++++++++++- tests/test_clean_agent_preview.py | 17 +++++++- 2 files changed, 79 insertions(+), 2 deletions(-) diff --git a/src/clean_agent_preview.py b/src/clean_agent_preview.py index 8d21d30aa..14db85bee 100644 --- a/src/clean_agent_preview.py +++ b/src/clean_agent_preview.py @@ -1239,6 +1239,32 @@ def prior_web_source_answer(user_text, history): return candidates[-1] if candidates else '' +def bounded_web_evidence_answer(user_text, source_links): + """Preserve useful Web evidence when a model will not stop searching. + + This is a last-resort terminal response, not a substitute for synthesis. It + deliberately reports only source titles and URLs already returned by the + search provider so the harness cannot invent a summary or discard evidence + behind a generic tool-loop error. + """ + unique = [] + for item in source_links or (): + value = str(item or '').strip() + if value and value not in unique: + unique.append(value) + if not unique: + return '' + subject = re.sub(r'\s+', ' ', str(user_text or '')).strip().rstrip('?.!') + return ( + f'I found current Web sources for “{subject}”, but could not complete a ' + 'reliable synthesis because the model kept requesting additional searches ' + 'after the bounded research budget. Here are the sources already found:\n\n' + + '\n'.join(f'- {link}' for link in unique[:5]) + + '\n\nThese are preliminary search results; open the strongest source or ask me to ' + 'retry the synthesis before relying on details not visible in the titles.' + ) + + def document_suggestions_event(result, *, failed=False): """Return the browser-owned inline-suggestion event for a successful call.""" if failed or not isinstance(result, dict): @@ -3815,6 +3841,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac artifact_body_handoff_target = '' artifact_write_phase = False suppression_completion_attempted = False + search_completion_attempted = False budget_completion_attempted = False answer_recovery_attempts = 0 force_no_tools_next_round = False @@ -4390,6 +4417,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac break terminal_denial = False terminal_suppression_violation = False + terminal_search_budget_violation = False terminal_budget_violation = False structured_terminal_response = '' round_recovery_messages = [] @@ -4450,7 +4478,7 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac if not native_workspace_enabled and web_search_attempts > 3: suppressed_tool_until_round['web_search'] = round_limit + 1 force_no_tools_next_round = True - terminal_suppression_violation = True + terminal_search_budget_violation = True calls += 1 raise ValueError( 'The bounded search-attempt budget is exhausted. Do not search ' @@ -5344,6 +5372,40 @@ async def stream_preview(*, endpoint_url, model, messages, headers, turn_contrac history.append({'role': 'assistant', 'content': refusal}) yield event({'type': 'final_response', 'content': refusal}) break + if terminal_search_budget_violation: + if not search_completion_attempted and round_number < round_limit: + search_completion_attempted = True + force_no_tools_next_round = True + recovery = ( + 'Research budget reached: no more tools will be offered. Using only ' + 'the search evidence already returned, provide the complete final ' + 'answer now with useful detail and source URLs. Do not emit a tool call.' + ) + if history and history[-1].get('_harness_control'): + history[-1]['content'] = ( + str(history[-1].get('content') or '') + ' ' + recovery + ) + else: + history.append({ + 'role': 'user', '_harness_control': True, 'content': recovery, + }) + yield event({ + 'type': 'completion_recovery', + 'reason': 'bounded_search_final_synthesis', + }) + continue + evidence_answer = bounded_web_evidence_answer( + direct_user_text, discovered_web_sources, + ) + if not evidence_answer: + evidence_answer = ( + 'I could not find usable Web evidence within the bounded search ' + 'attempts. I did not infer an answer from unsupported results. Try a ' + 'narrower topic, date range, organization, or source type.' + ) + history.append({'role': 'assistant', 'content': evidence_answer}) + yield event({'type': 'final_response', 'content': evidence_answer}) + break if terminal_suppression_violation: missing_artifacts = missing_workspace_artifacts(latest_user, workspace) if ( diff --git a/tests/test_clean_agent_preview.py b/tests/test_clean_agent_preview.py index f54cfca0e..7846e686b 100644 --- a/tests/test_clean_agent_preview.py +++ b/tests/test_clean_agent_preview.py @@ -5,7 +5,7 @@ import jsonschema import pytest import re -from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, web_fetch_observation_is_boilerplate, broad_current_web_request, record_tool_execution, align_structured_tool_history, provider_request_messages, offered_tool_alias, dependent_write_prerequisite_error, bounded_research_tool_policy, retrieved_source_urls, serialize_required_email_attachment_chain +from src.clean_agent_preview import conversation, readonly_call, preview_call_allowed, evaluate_preview_call, authorized_write_families, compact_schemas, normalize_preview_function_args, normalize_preview_call_args, private_browser_dom_batch, private_browser_state_transition, private_browser_success_repeat_limit, stream_preview, denied_response, execution_has_write_effect, requests_mutation, claims_completion, recent_successful_write_families, scope_preview_contract, multimodal_image_count, attachment_reference_count, active_document_context_message, active_email_context_message, targets_active_editor, active_editor_whole_draft_request, active_editor_suggestion_request, scope_active_editor_contract, native_execution_limits, interactive_execution_limit, runtime_required_artifacts, document_suggestions_event, document_suggestion_quality_error, required_read_tool_choice, required_active_editor_tool_choice, sealed_read_arguments, email_identifier_error, requested_item_limit, contract_item_limit, notes_terminal_response, documents_terminal_response, shell_listing_terminal_response, shell_output_terminal_response, ui_panel_terminal_response, ui_toggle_state_result, calendar_terminal_response, memory_terminal_response, tasks_terminal_response, task_list_requires_synthesis, skills_terminal_response, cookbook_servers_terminal_response, prior_short_answer_for_no_tool_summary, prior_collection_repeat_answer, prior_failed_operation_answer, prior_cookbook_server_answer, prior_workspace_path_answer, prior_web_source_answer, bounded_web_evidence_answer, inherit_referential_read_arguments, normalized_search_intent, requested_web_source_links, web_source_links, requested_web_link_limit, preserve_requested_web_recency, ground_referenced_note_content, note_search_result_empty, note_referent_error, research_referent_error, private_browser_open_url, private_browser_effective_url, web_fetch_observation_is_boilerplate, broad_current_web_request, record_tool_execution, align_structured_tool_history, provider_request_messages, offered_tool_alias, dependent_write_prerequisite_error, bounded_research_tool_policy, retrieved_source_urls, serialize_required_email_attachment_chain from src.tool_capabilities import capabilities_for_tool @@ -1891,6 +1891,21 @@ def test_broad_briefing_requires_substance_and_clickable_source_links(): assert not incomplete_broad_web_answer('Short answer.', 'What is Python?') +def test_bounded_web_evidence_answer_preserves_sources_without_claiming_synthesis(): + answer = bounded_web_evidence_answer( + "What's happening in Norway?", + [ + '[Source: Norway election update](https://example.org/norway-election)', + '[Source: Norway economy update](https://example.net/norway-economy)', + ], + ) + + assert 'could not complete a reliable synthesis' in answer + assert 'https://example.org/norway-election' in answer + assert 'https://example.net/norway-economy' in answer + assert 'repeated a tool call after that tool was disabled' not in answer + + def test_followup_search_must_change_subject_angle_not_only_freshness(): from src.clean_agent_preview import repeated_search_refinement