mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-08-05 02:45:28 +00:00
Merge 7dca3f801e into 20e7fc0164
This commit is contained in:
commit
a9e8ff7d19
6 changed files with 166 additions and 12 deletions
|
|
@ -9,6 +9,7 @@ from src.chat_helpers import extract_urls
|
|||
from src.youtube_handler import is_youtube_url
|
||||
from src.search import comprehensive_web_search, fetch_webpage_content
|
||||
from src.prompt_security import UNTRUSTED_CONTEXT_POLICY, untrusted_context_message
|
||||
from src.markitdown_runtime import original_filename
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
|
@ -368,17 +369,35 @@ class ChatProcessor:
|
|||
relevant = [r for r in results if r.get("similarity", 0) >= self.RAG_SIMILARITY_THRESHOLD]
|
||||
if relevant:
|
||||
logger.info(f"RAG: {len(relevant)}/{len(results)} results above threshold {self.RAG_SIMILARITY_THRESHOLD}")
|
||||
rag_sources = [
|
||||
{
|
||||
"filename": r["metadata"].get("filename", r["metadata"].get("source", "unknown")),
|
||||
rag_sources = []
|
||||
for r in relevant:
|
||||
meta = r.get("metadata") or {}
|
||||
src = {
|
||||
# Show the original document, not the internal
|
||||
# ``.md`` markitdown conversion name (issue #5666).
|
||||
"filename": original_filename(
|
||||
meta.get("filename", meta.get("source", "unknown"))
|
||||
),
|
||||
"snippet": r["document"][:200],
|
||||
"similarity": round(r.get("similarity", 0), 3)
|
||||
"similarity": round(r.get("similarity", 0), 3),
|
||||
}
|
||||
for r in relevant
|
||||
]
|
||||
rag_content = "Relevant documents:\n\n" + "\n\n---\n\n".join(
|
||||
f"[{s['filename']}]\n{r['document']}" for s, r in zip(rag_sources, relevant)
|
||||
)
|
||||
# Provenance tags — surfaced as chips in the UI and
|
||||
# woven into the injected context below so the model
|
||||
# can attribute snippets. Only present for KBs that
|
||||
# tag documents; absent keys render exactly as before.
|
||||
if meta.get("project"):
|
||||
src["project"] = meta["project"]
|
||||
if meta.get("org"):
|
||||
src["org"] = meta["org"]
|
||||
rag_sources.append(src)
|
||||
rag_parts = []
|
||||
for s, r in zip(rag_sources, relevant):
|
||||
prov = ", ".join(
|
||||
f"{k}: {s[k]}" for k in ("project", "org") if s.get(k)
|
||||
)
|
||||
header = f"{s['filename']} ({prov})" if prov else s["filename"]
|
||||
rag_parts.append(f"[{header}]\n{r['document']}")
|
||||
rag_content = "Relevant documents:\n\n" + "\n\n---\n\n".join(rag_parts)
|
||||
if len(rag_content) > 10000:
|
||||
rag_content = rag_content[:10000] + "\n[Truncated]"
|
||||
preface.append(untrusted_context_message("retrieved documents", rag_content))
|
||||
|
|
|
|||
|
|
@ -31,6 +31,25 @@ def is_markitdown_format(path: str) -> bool:
|
|||
return os.path.splitext(path)[1].lower() in MARKITDOWN_EXTS
|
||||
|
||||
|
||||
def original_filename(filename: str) -> str:
|
||||
"""Undo the ``.md`` suffix markitdown adds when converting Office files.
|
||||
|
||||
Office/EPUB documents land in a KB as ``<name>.<ext>.md`` (e.g.
|
||||
``Deck.pptx.md``) once converted. For a sources bibliography we want to
|
||||
cite the document the user actually ingested, so strip the trailing
|
||||
``.md`` when the remaining stem still carries a converted extension:
|
||||
``Deck.pptx.md`` -> ``Deck.pptx``. A hand-authored ``notes.md`` is left
|
||||
untouched — its stem (``notes``) has no converted extension. Non-str
|
||||
input is returned unchanged.
|
||||
"""
|
||||
if not isinstance(filename, str) or not filename.lower().endswith(".md"):
|
||||
return filename
|
||||
stem = filename[:-3]
|
||||
if os.path.splitext(stem)[1].lower() in MARKITDOWN_EXTS:
|
||||
return stem
|
||||
return filename
|
||||
|
||||
|
||||
def load_markitdown():
|
||||
"""Return the MarkItDown class, or raise a user-facing setup hint."""
|
||||
try:
|
||||
|
|
|
|||
|
|
@ -3623,7 +3623,10 @@ import { wireArrowUpRecall, getUserMessagesFromChatHistory } from './composerArr
|
|||
const item = document.createElement('div');
|
||||
item.className = 'rag-source-item';
|
||||
const _esc = uiModule.esc;
|
||||
item.innerHTML = `<strong>${_esc(src.filename)}</strong> <span class="rag-similarity">${(src.similarity * 100).toFixed(1)}%</span><div class="rag-snippet">${_esc(src.snippet)}</div>`;
|
||||
// Provenance chips (project/org) render only when tagged (#5666).
|
||||
const _tags = [src.project, src.org].filter(Boolean)
|
||||
.map(t => `<span class="rag-source-tag">${_esc(t)}</span>`).join('');
|
||||
item.innerHTML = `<strong>${_esc(src.filename)}</strong> <span class="rag-similarity">${(src.similarity * 100).toFixed(1)}%</span>${_tags}<div class="rag-snippet">${_esc(src.snippet)}</div>`;
|
||||
details.appendChild(item);
|
||||
});
|
||||
holder.querySelector('.body').appendChild(details);
|
||||
|
|
|
|||
|
|
@ -978,8 +978,9 @@ export function buildSourcesBox(sources, type, expanded) {
|
|||
/**
|
||||
* Build the RAG "Sources (N documents)" box — mirrors the live render in
|
||||
* chat.js so persisted rag_sources survive a refresh. Items carry a
|
||||
* filename, similarity %, and snippet (not URLs, unlike web sources).
|
||||
* @param {Array<{filename, similarity, snippet}>} sources
|
||||
* filename, similarity %, snippet, and optional project/org provenance tags
|
||||
* (not URLs, unlike web sources).
|
||||
* @param {Array<{filename, similarity, snippet, project?, org?}>} sources
|
||||
*/
|
||||
export function buildRagSourcesBox(sources) {
|
||||
if (!sources || !sources.length) return '';
|
||||
|
|
@ -988,8 +989,13 @@ export function buildRagSourcesBox(sources) {
|
|||
for (var i = 0; i < sources.length; i++) {
|
||||
var s = sources[i] || {};
|
||||
var pct = (typeof s.similarity === 'number') ? (s.similarity * 100).toFixed(1) + '%' : '';
|
||||
// Provenance chips (project/org) render only when tagged (#5666).
|
||||
var tags = '';
|
||||
if (s.project) tags += '<span class="rag-source-tag">' + esc(s.project) + '</span>';
|
||||
if (s.org) tags += '<span class="rag-source-tag">' + esc(s.org) + '</span>';
|
||||
items += '<div class="rag-source-item"><strong>' + esc(s.filename || '') + '</strong>'
|
||||
+ (pct ? ' <span class="rag-similarity">' + pct + '</span>' : '')
|
||||
+ tags
|
||||
+ '<div class="rag-snippet">' + esc(s.snippet || '') + '</div></div>';
|
||||
}
|
||||
return '<details class="rag-sources"><summary>Sources (' + sources.length + ' documents)</summary>' + items + '</details>';
|
||||
|
|
|
|||
|
|
@ -2374,6 +2374,20 @@ body.bg-pattern-sparkles {
|
|||
max-height: 60px;
|
||||
overflow: hidden;
|
||||
}
|
||||
/* Provenance chips (project/org) for a RAG source — reuse the neutral
|
||||
--fg mixes the surrounding .rag-* styles use, so no new palette. */
|
||||
.rag-source-tag {
|
||||
display: inline-block;
|
||||
margin-left: 6px;
|
||||
padding: 0 6px;
|
||||
font-size: 10px;
|
||||
line-height: 16px;
|
||||
border-radius: 8px;
|
||||
vertical-align: middle;
|
||||
color: color-mix(in srgb, var(--fg) 65%, transparent);
|
||||
background: color-mix(in srgb, var(--fg) 8%, transparent);
|
||||
border: 1px solid color-mix(in srgb, var(--fg) 12%, transparent);
|
||||
}
|
||||
.rag-file-delete {
|
||||
background: none;
|
||||
border: 1px solid var(--border);
|
||||
|
|
|
|||
93
tests/test_rag_source_bibliography.py
Normal file
93
tests/test_rag_source_bibliography.py
Normal file
|
|
@ -0,0 +1,93 @@
|
|||
"""Regression tests for issue #5666 — RAG sources bibliography.
|
||||
|
||||
Covers two behaviours the chat sources block gained:
|
||||
1. Converted Office files are cited by their original name (``Deck.pptx``),
|
||||
not the internal ``Deck.pptx.md`` markitdown conversion name.
|
||||
2. ``project`` / ``org`` provenance tags flow into both the ``rag_sources``
|
||||
list (rendered as chips) and the injected retrieval context — and are
|
||||
omitted, byte-for-byte as before, when a document carries no such tags.
|
||||
"""
|
||||
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from src.chat_processor import ChatProcessor
|
||||
from src.markitdown_runtime import original_filename
|
||||
|
||||
|
||||
def test_original_filename_strips_converted_office_suffix():
|
||||
assert original_filename("Deck.pptx.md") == "Deck.pptx"
|
||||
assert original_filename("Report.docx.md") == "Report.docx"
|
||||
assert original_filename("Sheet.XLSX.md") == "Sheet.XLSX" # case-insensitive
|
||||
|
||||
|
||||
def test_original_filename_leaves_plain_markdown_and_others_untouched():
|
||||
# A hand-authored markdown file's stem has no converted extension.
|
||||
assert original_filename("notes.md") == "notes.md"
|
||||
assert original_filename("README.md") == "README.md"
|
||||
# Non-.md files and non-str input pass through unchanged.
|
||||
assert original_filename("photo.png") == "photo.png"
|
||||
assert original_filename(None) is None
|
||||
|
||||
|
||||
def _processor_with_rag(hits):
|
||||
"""A ChatProcessor whose rag_manager.search returns ``hits``."""
|
||||
pdm = MagicMock()
|
||||
pdm.rag_manager.search.return_value = hits
|
||||
return ChatProcessor(memory_manager=MagicMock(), personal_docs_manager=pdm)
|
||||
|
||||
|
||||
def _preface_for(hits):
|
||||
processor = _processor_with_rag(hits)
|
||||
session = SimpleNamespace(endpoint_url="http://local", model="test", headers={})
|
||||
return processor.build_context_preface(
|
||||
message="What is the roadmap?",
|
||||
session=session,
|
||||
use_web=False,
|
||||
use_rag=True,
|
||||
use_memory=False,
|
||||
use_skills=False,
|
||||
)
|
||||
|
||||
|
||||
def test_rag_sources_surface_original_name_and_provenance_tags():
|
||||
hits = [{
|
||||
"document": "Q3 roadmap: ship the ingest pipeline.",
|
||||
"metadata": {
|
||||
"filename": "Atlas-Q3-Product-Roadmap.pptx.md",
|
||||
"project": "AI Platform",
|
||||
"org": "techinnovators",
|
||||
},
|
||||
"similarity": 0.82,
|
||||
}]
|
||||
preface, rag_sources, _web = _preface_for(hits)
|
||||
|
||||
assert len(rag_sources) == 1
|
||||
src = rag_sources[0]
|
||||
assert src["filename"] == "Atlas-Q3-Product-Roadmap.pptx" # .md stripped
|
||||
assert src["project"] == "AI Platform"
|
||||
assert src["org"] == "techinnovators"
|
||||
|
||||
# Provenance is woven into the injected retrieval context so the model
|
||||
# can attribute the snippet.
|
||||
injected = "\n".join(m["content"] for m in preface)
|
||||
assert "Atlas-Q3-Product-Roadmap.pptx (project: AI Platform, org: techinnovators)" in injected
|
||||
|
||||
|
||||
def test_rag_sources_without_tags_are_backwards_compatible():
|
||||
hits = [{
|
||||
"document": "Plain text note body.",
|
||||
"metadata": {"filename": "notes.md"},
|
||||
"similarity": 0.5,
|
||||
}]
|
||||
preface, rag_sources, _web = _preface_for(hits)
|
||||
|
||||
src = rag_sources[0]
|
||||
assert src["filename"] == "notes.md" # untouched
|
||||
# No provenance keys leak in when the document isn't tagged.
|
||||
assert "project" not in src
|
||||
assert "org" not in src
|
||||
|
||||
injected = "\n".join(m["content"] for m in preface)
|
||||
assert "[notes.md]" in injected # bare header, exactly as before
|
||||
assert "project:" not in injected
|
||||
Loading…
Add table
Reference in a new issue