This commit is contained in:
Wes Huber 2026-08-04 15:27:33 +02:00 committed by GitHub
commit a9e8ff7d19
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
6 changed files with 166 additions and 12 deletions

View file

@ -9,6 +9,7 @@ from src.chat_helpers import extract_urls
from src.youtube_handler import is_youtube_url
from src.search import comprehensive_web_search, fetch_webpage_content
from src.prompt_security import UNTRUSTED_CONTEXT_POLICY, untrusted_context_message
from src.markitdown_runtime import original_filename
logger = logging.getLogger(__name__)
@ -368,17 +369,35 @@ class ChatProcessor:
relevant = [r for r in results if r.get("similarity", 0) >= self.RAG_SIMILARITY_THRESHOLD]
if relevant:
logger.info(f"RAG: {len(relevant)}/{len(results)} results above threshold {self.RAG_SIMILARITY_THRESHOLD}")
rag_sources = [
{
"filename": r["metadata"].get("filename", r["metadata"].get("source", "unknown")),
rag_sources = []
for r in relevant:
meta = r.get("metadata") or {}
src = {
# Show the original document, not the internal
# ``.md`` markitdown conversion name (issue #5666).
"filename": original_filename(
meta.get("filename", meta.get("source", "unknown"))
),
"snippet": r["document"][:200],
"similarity": round(r.get("similarity", 0), 3)
"similarity": round(r.get("similarity", 0), 3),
}
for r in relevant
]
rag_content = "Relevant documents:\n\n" + "\n\n---\n\n".join(
f"[{s['filename']}]\n{r['document']}" for s, r in zip(rag_sources, relevant)
)
# Provenance tags — surfaced as chips in the UI and
# woven into the injected context below so the model
# can attribute snippets. Only present for KBs that
# tag documents; absent keys render exactly as before.
if meta.get("project"):
src["project"] = meta["project"]
if meta.get("org"):
src["org"] = meta["org"]
rag_sources.append(src)
rag_parts = []
for s, r in zip(rag_sources, relevant):
prov = ", ".join(
f"{k}: {s[k]}" for k in ("project", "org") if s.get(k)
)
header = f"{s['filename']} ({prov})" if prov else s["filename"]
rag_parts.append(f"[{header}]\n{r['document']}")
rag_content = "Relevant documents:\n\n" + "\n\n---\n\n".join(rag_parts)
if len(rag_content) > 10000:
rag_content = rag_content[:10000] + "\n[Truncated]"
preface.append(untrusted_context_message("retrieved documents", rag_content))

View file

@ -31,6 +31,25 @@ def is_markitdown_format(path: str) -> bool:
return os.path.splitext(path)[1].lower() in MARKITDOWN_EXTS
def original_filename(filename: str) -> str:
"""Undo the ``.md`` suffix markitdown adds when converting Office files.
Office/EPUB documents land in a KB as ``<name>.<ext>.md`` (e.g.
``Deck.pptx.md``) once converted. For a sources bibliography we want to
cite the document the user actually ingested, so strip the trailing
``.md`` when the remaining stem still carries a converted extension:
``Deck.pptx.md`` -> ``Deck.pptx``. A hand-authored ``notes.md`` is left
untouched its stem (``notes``) has no converted extension. Non-str
input is returned unchanged.
"""
if not isinstance(filename, str) or not filename.lower().endswith(".md"):
return filename
stem = filename[:-3]
if os.path.splitext(stem)[1].lower() in MARKITDOWN_EXTS:
return stem
return filename
def load_markitdown():
"""Return the MarkItDown class, or raise a user-facing setup hint."""
try:

View file

@ -3623,7 +3623,10 @@ import { wireArrowUpRecall, getUserMessagesFromChatHistory } from './composerArr
const item = document.createElement('div');
item.className = 'rag-source-item';
const _esc = uiModule.esc;
item.innerHTML = `<strong>${_esc(src.filename)}</strong> <span class="rag-similarity">${(src.similarity * 100).toFixed(1)}%</span><div class="rag-snippet">${_esc(src.snippet)}</div>`;
// Provenance chips (project/org) render only when tagged (#5666).
const _tags = [src.project, src.org].filter(Boolean)
.map(t => `<span class="rag-source-tag">${_esc(t)}</span>`).join('');
item.innerHTML = `<strong>${_esc(src.filename)}</strong> <span class="rag-similarity">${(src.similarity * 100).toFixed(1)}%</span>${_tags}<div class="rag-snippet">${_esc(src.snippet)}</div>`;
details.appendChild(item);
});
holder.querySelector('.body').appendChild(details);

View file

@ -978,8 +978,9 @@ export function buildSourcesBox(sources, type, expanded) {
/**
* Build the RAG "Sources (N documents)" box mirrors the live render in
* chat.js so persisted rag_sources survive a refresh. Items carry a
* filename, similarity %, and snippet (not URLs, unlike web sources).
* @param {Array<{filename, similarity, snippet}>} sources
* filename, similarity %, snippet, and optional project/org provenance tags
* (not URLs, unlike web sources).
* @param {Array<{filename, similarity, snippet, project?, org?}>} sources
*/
export function buildRagSourcesBox(sources) {
if (!sources || !sources.length) return '';
@ -988,8 +989,13 @@ export function buildRagSourcesBox(sources) {
for (var i = 0; i < sources.length; i++) {
var s = sources[i] || {};
var pct = (typeof s.similarity === 'number') ? (s.similarity * 100).toFixed(1) + '%' : '';
// Provenance chips (project/org) render only when tagged (#5666).
var tags = '';
if (s.project) tags += '<span class="rag-source-tag">' + esc(s.project) + '</span>';
if (s.org) tags += '<span class="rag-source-tag">' + esc(s.org) + '</span>';
items += '<div class="rag-source-item"><strong>' + esc(s.filename || '') + '</strong>'
+ (pct ? ' <span class="rag-similarity">' + pct + '</span>' : '')
+ tags
+ '<div class="rag-snippet">' + esc(s.snippet || '') + '</div></div>';
}
return '<details class="rag-sources"><summary>Sources (' + sources.length + ' documents)</summary>' + items + '</details>';

View file

@ -2374,6 +2374,20 @@ body.bg-pattern-sparkles {
max-height: 60px;
overflow: hidden;
}
/* Provenance chips (project/org) for a RAG source reuse the neutral
--fg mixes the surrounding .rag-* styles use, so no new palette. */
.rag-source-tag {
display: inline-block;
margin-left: 6px;
padding: 0 6px;
font-size: 10px;
line-height: 16px;
border-radius: 8px;
vertical-align: middle;
color: color-mix(in srgb, var(--fg) 65%, transparent);
background: color-mix(in srgb, var(--fg) 8%, transparent);
border: 1px solid color-mix(in srgb, var(--fg) 12%, transparent);
}
.rag-file-delete {
background: none;
border: 1px solid var(--border);

View file

@ -0,0 +1,93 @@
"""Regression tests for issue #5666 — RAG sources bibliography.
Covers two behaviours the chat sources block gained:
1. Converted Office files are cited by their original name (``Deck.pptx``),
not the internal ``Deck.pptx.md`` markitdown conversion name.
2. ``project`` / ``org`` provenance tags flow into both the ``rag_sources``
list (rendered as chips) and the injected retrieval context and are
omitted, byte-for-byte as before, when a document carries no such tags.
"""
from types import SimpleNamespace
from unittest.mock import MagicMock
from src.chat_processor import ChatProcessor
from src.markitdown_runtime import original_filename
def test_original_filename_strips_converted_office_suffix():
assert original_filename("Deck.pptx.md") == "Deck.pptx"
assert original_filename("Report.docx.md") == "Report.docx"
assert original_filename("Sheet.XLSX.md") == "Sheet.XLSX" # case-insensitive
def test_original_filename_leaves_plain_markdown_and_others_untouched():
# A hand-authored markdown file's stem has no converted extension.
assert original_filename("notes.md") == "notes.md"
assert original_filename("README.md") == "README.md"
# Non-.md files and non-str input pass through unchanged.
assert original_filename("photo.png") == "photo.png"
assert original_filename(None) is None
def _processor_with_rag(hits):
"""A ChatProcessor whose rag_manager.search returns ``hits``."""
pdm = MagicMock()
pdm.rag_manager.search.return_value = hits
return ChatProcessor(memory_manager=MagicMock(), personal_docs_manager=pdm)
def _preface_for(hits):
processor = _processor_with_rag(hits)
session = SimpleNamespace(endpoint_url="http://local", model="test", headers={})
return processor.build_context_preface(
message="What is the roadmap?",
session=session,
use_web=False,
use_rag=True,
use_memory=False,
use_skills=False,
)
def test_rag_sources_surface_original_name_and_provenance_tags():
hits = [{
"document": "Q3 roadmap: ship the ingest pipeline.",
"metadata": {
"filename": "Atlas-Q3-Product-Roadmap.pptx.md",
"project": "AI Platform",
"org": "techinnovators",
},
"similarity": 0.82,
}]
preface, rag_sources, _web = _preface_for(hits)
assert len(rag_sources) == 1
src = rag_sources[0]
assert src["filename"] == "Atlas-Q3-Product-Roadmap.pptx" # .md stripped
assert src["project"] == "AI Platform"
assert src["org"] == "techinnovators"
# Provenance is woven into the injected retrieval context so the model
# can attribute the snippet.
injected = "\n".join(m["content"] for m in preface)
assert "Atlas-Q3-Product-Roadmap.pptx (project: AI Platform, org: techinnovators)" in injected
def test_rag_sources_without_tags_are_backwards_compatible():
hits = [{
"document": "Plain text note body.",
"metadata": {"filename": "notes.md"},
"similarity": 0.5,
}]
preface, rag_sources, _web = _preface_for(hits)
src = rag_sources[0]
assert src["filename"] == "notes.md" # untouched
# No provenance keys leak in when the document isn't tagged.
assert "project" not in src
assert "org" not in src
injected = "\n".join(m["content"] for m in preface)
assert "[notes.md]" in injected # bare header, exactly as before
assert "project:" not in injected