from __future__ import annotations import sqlite3 from pathlib import Path from evaluation.rag_metrics import ( expected_phrase_matches, ) from scripts.rag_query_evidence import ( find_top_query_evidence_chunks, ) def create_db( tmp_path: Path, ) -> Path: db = ( tmp_path / "e5_1.sqlite" ) with sqlite3.connect( db ) as conn: conn.execute( """ CREATE TABLE chunks ( id INTEGER PRIMARY KEY, chunk_id TEXT NOT NULL, document_path TEXT NOT NULL, published INTEGER NOT NULL, chunk_index INTEGER NOT NULL, heading_paths_json TEXT NOT NULL, text TEXT NOT NULL ) """ ) return db def add_chunk( db: Path, *, path: str, index: int, text: str, heading_paths_json: str = "[]", ) -> None: with sqlite3.connect( db ) as conn: conn.execute( """ INSERT INTO chunks ( chunk_id, document_path, published, chunk_index, heading_paths_json, text ) VALUES (?, ?, 1, ?, ?, ?) """, ( f"{path}::chunk-{index}", path, index, heading_paths_json, text, ), ) def best_text( db: Path, *, path: str, query: str, ) -> str: with sqlite3.connect( db ) as conn: conn.row_factory = ( sqlite3.Row ) chunks = ( find_top_query_evidence_chunks( conn, path, query, published_only=True, top_k=4, ) ) assert chunks return str( chunks[ 0 ].get( "focus_text" ) or "" ) def test_title_question_prefers_explicit_dp_title_proposal( tmp_path: Path, ) -> None: db = create_db( tmp_path ) path = ( "pages/students/2016/" "test_student/README.md" ) add_chunk( db, path=path, index=0, text=( "# Test Student\n\n" "Návrh na názov DP:\n" "Anotácia a rozpoznávanie " "pomenovaných entít " "v slovenskom jazyku." ), ) add_chunk( db, path=path, index=1, heading_paths_json=( '[["Diplomová práca 2021"]]' ), text=( "## Diplomová práca 2021\n\n" "Stav: anotovanie dát " "a experimenty." ), ) text = best_text( db, path=path, query=( "Aký je názov diplomovej " "práce osoby Test Student?" ), ) assert ( "Anotácia a rozpoznávanie" in text ) def test_goal_question_prefers_english_goals_block( tmp_path: Path, ) -> None: db = create_db( tmp_path ) path = ( "pages/interns/" "test_intern/README.md" ) add_chunk( db, path=path, index=0, text=( "Test Intern. " "Summer internship." ), ) add_chunk( db, path=path, index=1, heading_paths_json=( '[["Goals"]]' ), text=( "## Goals\n" "- Be able to recognize " "unknown named entities\n" "- Create a manually annotated " "training set\n" "- Propose an annotation schema" ), ) text = best_text( db, path=path, query=( "Aké tri hlavné ciele " "mal Test Intern?" ), ) assert ( "recognize unknown named entities" in text ) assert ( "manually annotated training set" in text ) assert ( "annotation schema" in text ) def test_question_count_prefers_english_question_annotation_block( tmp_path: Path, ) -> None: db = create_db( tmp_path ) path = ( "pages/topics/question/" "README.md" ) add_chunk( db, path=path, index=0, text=( "Question Answering " "project description." ), ) add_chunk( db, path=path, index=6, heading_paths_json=( '[["Question Annotation"]]' ), text=( "### Question Annotation\n" "Input: A set of paragraphs\n" "Output: 5 questions " "for each paragraph" ), ) text = best_text( db, path=path, query=( "Koľko otázok má vzniknúť " "pre každý odsek " "pri anotácii otázok?" ), ) assert ( "5 questions for each paragraph" in text ) def test_status_question_prefers_status_block_even_with_technology_names_only( tmp_path: Path, ) -> None: db = create_db( tmp_path ) path = ( "pages/students/2022/" "test_student/README.md" ) add_chunk( db, path=path, index=0, text=( "Diplomový projekt " "o znalostnom grafe " "a GraphRAG." ), ) add_chunk( db, path=path, index=4, heading_paths_json=( '[["Stav"]]' ), text=( "## Stav\n" "- SQLite\n" "- FastAPI" ), ) text = best_text( db, path=path, query=( "Akú databázu a backend už " "Test Student podľa stavu používal?" ), ) assert ( "SQLite" in text ) assert ( "FastAPI" in text ) def test_ner_matches_named_entities_semantically() -> None: assert expected_phrase_matches( "NER", ( "Dokumenty sa venujú " "rozpoznávaniu pomenovaných " "entít a ich anotácii." ), ) def test_numeric_match_remains_strict() -> None: assert expected_phrase_matches( "2021", ( "Práca bola " "v roku 2021." ), ) assert not expected_phrase_matches( "2021", ( "Identifikátor " "je 20210." ), )