365 lines
6.5 KiB
Python
365 lines
6.5 KiB
Python
from __future__ import annotations
|
|
|
|
import sqlite3
|
|
from pathlib import Path
|
|
|
|
from evaluation.rag_metrics import (
|
|
expected_phrase_matches,
|
|
)
|
|
from scripts.rag_query_evidence import (
|
|
find_top_query_evidence_chunks,
|
|
)
|
|
|
|
|
|
def create_db(
|
|
tmp_path: Path,
|
|
) -> Path:
|
|
db = (
|
|
tmp_path
|
|
/ "e5_1.sqlite"
|
|
)
|
|
|
|
with sqlite3.connect(
|
|
db
|
|
) as conn:
|
|
conn.execute(
|
|
"""
|
|
CREATE TABLE chunks (
|
|
id INTEGER PRIMARY KEY,
|
|
chunk_id TEXT NOT NULL,
|
|
document_path TEXT NOT NULL,
|
|
published INTEGER NOT NULL,
|
|
chunk_index INTEGER NOT NULL,
|
|
heading_paths_json TEXT NOT NULL,
|
|
text TEXT NOT NULL
|
|
)
|
|
"""
|
|
)
|
|
|
|
return db
|
|
|
|
|
|
def add_chunk(
|
|
db: Path,
|
|
*,
|
|
path: str,
|
|
index: int,
|
|
text: str,
|
|
heading_paths_json: str = "[]",
|
|
) -> None:
|
|
with sqlite3.connect(
|
|
db
|
|
) as conn:
|
|
conn.execute(
|
|
"""
|
|
INSERT INTO chunks (
|
|
chunk_id,
|
|
document_path,
|
|
published,
|
|
chunk_index,
|
|
heading_paths_json,
|
|
text
|
|
)
|
|
VALUES (?, ?, 1, ?, ?, ?)
|
|
""",
|
|
(
|
|
f"{path}::chunk-{index}",
|
|
path,
|
|
index,
|
|
heading_paths_json,
|
|
text,
|
|
),
|
|
)
|
|
|
|
|
|
def best_text(
|
|
db: Path,
|
|
*,
|
|
path: str,
|
|
query: str,
|
|
) -> str:
|
|
with sqlite3.connect(
|
|
db
|
|
) as conn:
|
|
conn.row_factory = (
|
|
sqlite3.Row
|
|
)
|
|
|
|
chunks = (
|
|
find_top_query_evidence_chunks(
|
|
conn,
|
|
path,
|
|
query,
|
|
published_only=True,
|
|
top_k=4,
|
|
)
|
|
)
|
|
|
|
assert chunks
|
|
|
|
return str(
|
|
chunks[
|
|
0
|
|
].get(
|
|
"focus_text"
|
|
)
|
|
or ""
|
|
)
|
|
|
|
|
|
def test_title_question_prefers_explicit_dp_title_proposal(
|
|
tmp_path: Path,
|
|
) -> None:
|
|
db = create_db(
|
|
tmp_path
|
|
)
|
|
|
|
path = (
|
|
"pages/students/2016/"
|
|
"test_student/README.md"
|
|
)
|
|
|
|
add_chunk(
|
|
db,
|
|
path=path,
|
|
index=0,
|
|
text=(
|
|
"# Test Student\n\n"
|
|
"Návrh na názov DP:\n"
|
|
"Anotácia a rozpoznávanie "
|
|
"pomenovaných entít "
|
|
"v slovenskom jazyku."
|
|
),
|
|
)
|
|
|
|
add_chunk(
|
|
db,
|
|
path=path,
|
|
index=1,
|
|
heading_paths_json=(
|
|
'[["Diplomová práca 2021"]]'
|
|
),
|
|
text=(
|
|
"## Diplomová práca 2021\n\n"
|
|
"Stav: anotovanie dát "
|
|
"a experimenty."
|
|
),
|
|
)
|
|
|
|
text = best_text(
|
|
db,
|
|
path=path,
|
|
query=(
|
|
"Aký je názov diplomovej "
|
|
"práce osoby Test Student?"
|
|
),
|
|
)
|
|
|
|
assert (
|
|
"Anotácia a rozpoznávanie"
|
|
in text
|
|
)
|
|
|
|
|
|
def test_goal_question_prefers_english_goals_block(
|
|
tmp_path: Path,
|
|
) -> None:
|
|
db = create_db(
|
|
tmp_path
|
|
)
|
|
|
|
path = (
|
|
"pages/interns/"
|
|
"test_intern/README.md"
|
|
)
|
|
|
|
add_chunk(
|
|
db,
|
|
path=path,
|
|
index=0,
|
|
text=(
|
|
"Test Intern. "
|
|
"Summer internship."
|
|
),
|
|
)
|
|
|
|
add_chunk(
|
|
db,
|
|
path=path,
|
|
index=1,
|
|
heading_paths_json=(
|
|
'[["Goals"]]'
|
|
),
|
|
text=(
|
|
"## Goals\n"
|
|
"- Be able to recognize "
|
|
"unknown named entities\n"
|
|
"- Create a manually annotated "
|
|
"training set\n"
|
|
"- Propose an annotation schema"
|
|
),
|
|
)
|
|
|
|
text = best_text(
|
|
db,
|
|
path=path,
|
|
query=(
|
|
"Aké tri hlavné ciele "
|
|
"mal Test Intern?"
|
|
),
|
|
)
|
|
|
|
assert (
|
|
"recognize unknown named entities"
|
|
in text
|
|
)
|
|
|
|
assert (
|
|
"manually annotated training set"
|
|
in text
|
|
)
|
|
|
|
assert (
|
|
"annotation schema"
|
|
in text
|
|
)
|
|
|
|
|
|
def test_question_count_prefers_english_question_annotation_block(
|
|
tmp_path: Path,
|
|
) -> None:
|
|
db = create_db(
|
|
tmp_path
|
|
)
|
|
|
|
path = (
|
|
"pages/topics/question/"
|
|
"README.md"
|
|
)
|
|
|
|
add_chunk(
|
|
db,
|
|
path=path,
|
|
index=0,
|
|
text=(
|
|
"Question Answering "
|
|
"project description."
|
|
),
|
|
)
|
|
|
|
add_chunk(
|
|
db,
|
|
path=path,
|
|
index=6,
|
|
heading_paths_json=(
|
|
'[["Question Annotation"]]'
|
|
),
|
|
text=(
|
|
"### Question Annotation\n"
|
|
"Input: A set of paragraphs\n"
|
|
"Output: 5 questions "
|
|
"for each paragraph"
|
|
),
|
|
)
|
|
|
|
text = best_text(
|
|
db,
|
|
path=path,
|
|
query=(
|
|
"Koľko otázok má vzniknúť "
|
|
"pre každý odsek "
|
|
"pri anotácii otázok?"
|
|
),
|
|
)
|
|
|
|
assert (
|
|
"5 questions for each paragraph"
|
|
in text
|
|
)
|
|
|
|
|
|
def test_status_question_prefers_status_block_even_with_technology_names_only(
|
|
tmp_path: Path,
|
|
) -> None:
|
|
db = create_db(
|
|
tmp_path
|
|
)
|
|
|
|
path = (
|
|
"pages/students/2022/"
|
|
"test_student/README.md"
|
|
)
|
|
|
|
add_chunk(
|
|
db,
|
|
path=path,
|
|
index=0,
|
|
text=(
|
|
"Diplomový projekt "
|
|
"o znalostnom grafe "
|
|
"a GraphRAG."
|
|
),
|
|
)
|
|
|
|
add_chunk(
|
|
db,
|
|
path=path,
|
|
index=4,
|
|
heading_paths_json=(
|
|
'[["Stav"]]'
|
|
),
|
|
text=(
|
|
"## Stav\n"
|
|
"- SQLite\n"
|
|
"- FastAPI"
|
|
),
|
|
)
|
|
|
|
text = best_text(
|
|
db,
|
|
path=path,
|
|
query=(
|
|
"Akú databázu a backend už "
|
|
"Test Student podľa stavu používal?"
|
|
),
|
|
)
|
|
|
|
assert (
|
|
"SQLite"
|
|
in text
|
|
)
|
|
|
|
assert (
|
|
"FastAPI"
|
|
in text
|
|
)
|
|
|
|
|
|
def test_ner_matches_named_entities_semantically() -> None:
|
|
assert expected_phrase_matches(
|
|
"NER",
|
|
(
|
|
"Dokumenty sa venujú "
|
|
"rozpoznávaniu pomenovaných "
|
|
"entít a ich anotácii."
|
|
),
|
|
)
|
|
|
|
|
|
def test_numeric_match_remains_strict() -> None:
|
|
assert expected_phrase_matches(
|
|
"2021",
|
|
(
|
|
"Práca bola "
|
|
"v roku 2021."
|
|
),
|
|
)
|
|
|
|
assert not expected_phrase_matches(
|
|
"2021",
|
|
(
|
|
"Identifikátor "
|
|
"je 20210."
|
|
),
|
|
)
|