dp-zp-agent/test/test_rag_query_evidence.py

522 lines
9.6 KiB
Python

from __future__ import annotations
import sqlite3
from pathlib import Path
from scripts.rag_query_evidence import (
build_focus_excerpt,
expand_results_with_query_evidence,
find_best_query_evidence_chunk,
)
def make_db(
tmp_path: Path,
) -> Path:
db = (
tmp_path
/ "evidence.sqlite"
)
with sqlite3.connect(
db
) as conn:
conn.execute(
"""
CREATE TABLE chunks (
id INTEGER PRIMARY KEY,
chunk_id TEXT UNIQUE NOT NULL,
document_path TEXT NOT NULL,
title TEXT,
author TEXT,
published INTEGER,
chunk_index INTEGER NOT NULL,
heading_paths_json TEXT
NOT NULL DEFAULT '[]',
text TEXT NOT NULL
)
"""
)
return db
def add_chunk(
db: Path,
*,
path: str,
index: int,
title: str,
heading: str,
text: str,
published: bool = True,
) -> None:
with sqlite3.connect(
db
) as conn:
conn.execute(
"""
INSERT INTO chunks (
chunk_id,
document_path,
title,
author,
published,
chunk_index,
heading_paths_json,
text
)
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
""",
(
f"{path}::chunk-{index}",
path,
title,
"Daniel Hladek",
(
1
if published
else 0
),
index,
heading,
text,
),
)
def result(
path: str,
index: int,
title: str,
text: str,
) -> dict:
return {
"chunk_id": (
f"{path}::chunk-{index}"
),
"document_path": path,
"title": title,
"author": "Daniel Hladek",
"published": True,
"chunk_index": index,
"heading_paths": [],
"text": text,
"source_url": (
"https://example.test/student"
),
"match_strategy": "hybrid",
"fts_rank": 1,
"vector_rank": 1,
"vector_score": 0.9,
"hybrid_score": 0.03,
}
def test_reverse_title_selects_tomas_2022_chunk(
tmp_path: Path,
) -> None:
db = make_db(
tmp_path
)
path = (
"pages/students/2016/"
"tomas_kucharik/README.md"
)
exact_title = (
"Tvorba korpusu otázok a odpovedí "
"v slovenskom jazyku pomocou "
"strojového prekladu"
)
add_chunk(
db,
path=path,
index=1,
title="Tomáš Kuchárik",
heading=(
'[["Tomáš Kuchárik", '
'"Diplomová práca 2022"]]'
),
text=(
"## Diplomová práca 2022\n\n"
f"Názov: {exact_title}"
),
)
add_chunk(
db,
path=path,
index=5,
title="Tomáš Kuchárik",
heading=(
'[["Tomáš Kuchárik", '
'"Diplomová práca 2021"]]'
),
text=(
"## Diplomová práca 2021\n\n"
"Názov: Tvorba korpusu otázok "
"a odpovedí v slovenskom jazyku "
"pomocou crowdsourcingu"
),
)
with sqlite3.connect(
db
) as conn:
conn.row_factory = (
sqlite3.Row
)
best = (
find_best_query_evidence_chunk(
conn,
path,
f"{exact_title} autor",
)
)
assert best is not None
assert (
best[
"chunk_index"
]
== 1
)
assert (
"strojového prekladu"
in best[
"text"
]
)
def test_expansion_preserves_retrieval_chunk_and_adds_2022_evidence(
tmp_path: Path,
) -> None:
db = make_db(
tmp_path
)
path = (
"pages/students/2016/"
"tomas_kucharik/README.md"
)
exact_title = (
"Tvorba korpusu otázok a odpovedí "
"v slovenskom jazyku pomocou "
"strojového prekladu"
)
add_chunk(
db,
path=path,
index=1,
title="Tomáš Kuchárik",
heading=(
'[["Tomáš Kuchárik", '
'"Diplomová práca 2022"]]'
),
text=(
f"Názov: {exact_title}"
),
)
add_chunk(
db,
path=path,
index=5,
title="Tomáš Kuchárik",
heading=(
'[["Tomáš Kuchárik", '
'"Diplomová práca 2021"]]'
),
text=(
"Názov: Tvorba korpusu "
"pomocou crowdsourcingu"
),
)
original = result(
path,
5,
"Tomáš Kuchárik",
"crowdsourcing",
)
expanded = (
expand_results_with_query_evidence(
db,
exact_title,
[
original
],
)
)
item = expanded[
0
]
assert (
item[
"chunk_id"
]
== original[
"chunk_id"
]
)
assert (
item[
"hybrid_score"
]
== original[
"hybrid_score"
]
)
assert (
item[
"query_evidence"
][
"applied"
]
is True
)
assert (
item[
"query_evidence"
][
"evidence_chunk_index"
]
== 1
)
assert (
"strojového prekladu"
in item[
"query_focus_text"
]
)
def test_us_steel_focus_contains_gnn_task() -> None:
text = (
"Stretnutie 1.10.\n"
"Stav:\n"
"- Štúdium základov neurónových sietí\n"
"- Úvodné stretnutie s US Steel\n"
"Úlohy:\n"
"- Vypracovať prehľad aktuálnych "
"metód grafových neurónových sietí\n"
"- Nájsť a vyskúšať toolkit na GNN.\n"
"- Naštudovať dáta z US Steel."
)
excerpt = (
build_focus_excerpt(
(
"Maroš Harahus US Steel "
"metódy študovať"
),
text,
)
)
assert (
"US Steel"
in excerpt
)
assert (
"grafových neurónových sietí"
in excerpt
)
assert (
"toolkit na GNN"
in excerpt
)
def test_maros_own_document_gets_gnn_evidence(
tmp_path: Path,
) -> None:
db = make_db(
tmp_path
)
path = (
"pages/students/2016/"
"maros_harahus/README.md"
)
text = (
"Úlohy:\n"
"- Vypracovať prehľad aktuálnych "
"metód grafových neurónových sietí\n"
"- Nájsť a vyskúšať toolkit na GNN."
)
add_chunk(
db,
path=path,
index=19,
title="Maroš Harahus",
heading=(
'[["Maroš Harahus", '
'"Prvý ročník PhD štúdia"]]'
),
text=text,
)
expanded = (
expand_results_with_query_evidence(
db,
(
"Maroš Harahus "
"grafové neurónové siete"
),
[
result(
path,
19,
"Maroš Harahus",
text,
)
],
)
)
assert (
expanded[
0
][
"query_evidence"
][
"applied"
]
is True
)
assert (
"grafových neurónových sietí"
in expanded[
0
][
"query_focus_text"
]
)
def test_published_only_ignores_unpublished_better_chunk(
tmp_path: Path,
) -> None:
db = make_db(
tmp_path
)
path = (
"pages/students/2016/"
"tomas_kucharik/README.md"
)
query = (
"Tvorba korpusu otázok a odpovedí "
"pomocou strojového prekladu"
)
add_chunk(
db,
path=path,
index=1,
title="Tomáš Kuchárik",
heading=(
'[["Diplomová práca 2022"]]'
),
text=(
"strojový preklad"
),
)
add_chunk(
db,
path=path,
index=2,
title="Tomáš Kuchárik",
heading=(
'[["Diplomová práca 2022"]]'
),
text=query,
published=False,
)
with sqlite3.connect(
db
) as conn:
conn.row_factory = (
sqlite3.Row
)
best = (
find_best_query_evidence_chunk(
conn,
path,
query,
published_only=True,
)
)
assert best is not None
assert (
best[
"chunk_index"
]
== 1
)
def test_missing_chunks_table_is_safe_noop(
tmp_path: Path,
) -> None:
db = (
tmp_path
/ "empty.sqlite"
)
with sqlite3.connect(
db
):
pass
original = result(
(
"pages/students/2016/"
"test/README.md"
),
0,
"Test",
"text",
)
assert (
expand_results_with_query_evidence(
db,
"nejaký dotaz",
[
original
],
)
== [
original
]
)