From 6080d001b53deb67bd2060895c0187ed1c9ce08a Mon Sep 17 00:00:00 2001 From: jp170na Date: Sun, 27 Sep 2026 14:57:00 +0200 Subject: [PATCH] pridanie topn testov query evidence --- test/test_rag_query_evidence_topn.py | 163 +++++++++++++++++++++++++++ 1 file changed, 163 insertions(+) create mode 100644 test/test_rag_query_evidence_topn.py diff --git a/test/test_rag_query_evidence_topn.py b/test/test_rag_query_evidence_topn.py new file mode 100644 index 0000000..12dfb4a --- /dev/null +++ b/test/test_rag_query_evidence_topn.py @@ -0,0 +1,163 @@ +from __future__ import annotations + +import sqlite3 +from pathlib import Path + +from scripts.rag_query_evidence import ( + expand_results_with_query_evidence, + find_top_query_evidence_chunks, +) + + +def create_db(tmp_path: Path) -> Path: + db = tmp_path / "evidence.sqlite" + + with sqlite3.connect(db) as conn: + conn.execute( + """ + CREATE TABLE chunks ( + id INTEGER PRIMARY KEY, + chunk_id TEXT, + document_path TEXT, + title TEXT, + author TEXT, + published INTEGER, + chunk_index INTEGER, + heading_paths_json TEXT, + text TEXT + ) + """ + ) + + return db + + +def insert_chunk( + db: Path, + *, + index: int, + text: str, + published: bool = True, +) -> None: + path = "pages/students/2022/jan_ptak/README.md" + + with sqlite3.connect(db) as conn: + conn.execute( + """ + INSERT INTO chunks( + chunk_id, + document_path, + title, + author, + published, + chunk_index, + heading_paths_json, + text + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?) + """, + ( + f"{path}::chunk-{index}", + path, + "Ján Pták", + "Daniel Hladek", + 1 if published else 0, + index, + "[]", + text, + ), + ) + + +def test_top_evidence_returns_more_than_one_relevant_chunk(tmp_path: Path) -> None: + db = create_db(tmp_path) + insert_chunk( + db, + index=0, + text="Ján Pták. Stav: databáza SQLite.", + ) + insert_chunk( + db, + index=1, + text="Ján Pták. Backend systému je FastAPI.", + ) + insert_chunk( + db, + index=2, + text="Ján Pták. Nesúvisiaci text o prezentácii.", + ) + + with sqlite3.connect(db) as conn: + conn.row_factory = sqlite3.Row + chunks = find_top_query_evidence_chunks( + conn, + "pages/students/2022/jan_ptak/README.md", + "Ján Pták databáza backend", + published_only=True, + top_k=3, + ) + + combined = "\n".join( + str(item.get("focus_text") or "") + for item in chunks + ) + + assert "SQLite" in combined + assert "FastAPI" in combined + assert len(chunks) >= 2 + + +def test_expansion_exposes_evidence_blocks_and_legacy_fields(tmp_path: Path) -> None: + db = create_db(tmp_path) + insert_chunk( + db, + index=0, + text="Matej Ščišľak vytvoril 110 otázok na testovanie systému.", + ) + path = "pages/students/2022/jan_ptak/README.md" + base = [ + { + "chunk_id": f"{path}::chunk-9", + "document_path": path, + "title": "Ján Pták", + "author": "Daniel Hladek", + "published": True, + "chunk_index": 9, + "heading_paths": [], + "text": "menej relevantný text", + } + ] + + result = expand_results_with_query_evidence( + db, + "Matej Ščišľak 110 otázok testovanie systému", + base, + published_only=True, + ) + + item = result[0] + assert item["query_evidence_blocks"] + assert "110" in item["query_focus_text"] + assert item["query_evidence"]["applied"] is True + assert item["query_evidence"]["evidence_chunks"] + + +def test_published_only_excludes_unpublished_evidence(tmp_path: Path) -> None: + db = create_db(tmp_path) + insert_chunk( + db, + index=0, + text="Ján Pták databáza SQLite backend FastAPI.", + published=False, + ) + + with sqlite3.connect(db) as conn: + conn.row_factory = sqlite3.Row + chunks = find_top_query_evidence_chunks( + conn, + "pages/students/2022/jan_ptak/README.md", + "Ján Pták databáza backend", + published_only=True, + top_k=3, + ) + + assert chunks == []