255 lines
7.0 KiB
Python
255 lines
7.0 KiB
Python
from __future__ import annotations
|
|
|
|
from evaluation.rag_metrics import (
|
|
evaluate_answer,
|
|
expected_phrase_matches,
|
|
)
|
|
|
|
|
|
def test_compact_unit_spacing_matches() -> None:
|
|
assert expected_phrase_matches(
|
|
"450MB",
|
|
"Nazbieraných bolo približne 450 MB textu.",
|
|
)
|
|
|
|
|
|
def test_llm_matches_large_language_model() -> None:
|
|
assert expected_phrase_matches(
|
|
"LLM",
|
|
"Projekt používa veľký jazykový model.",
|
|
)
|
|
|
|
|
|
def test_named_entities_matches_slovak_wording() -> None:
|
|
assert expected_phrase_matches(
|
|
"named entities",
|
|
"Stáž sa venovala pomenovaným entitám.",
|
|
)
|
|
|
|
|
|
def test_knowledge_graph_matches_slovak_wording() -> None:
|
|
assert expected_phrase_matches(
|
|
"knowledge graph",
|
|
"Cieľom bolo zostaviť znalostný graf.",
|
|
)
|
|
|
|
|
|
def test_multilingual_medical_triplet_phrase_matches_translation() -> None:
|
|
assert expected_phrase_matches(
|
|
"multilinguálna extrakcia trojíc z medicínskych dát",
|
|
"Téma bola viacjazyčné extrahovanie trojíc z lekárskych dát.",
|
|
)
|
|
|
|
|
|
def test_cesar_goal_matches_slovak_equivalent() -> None:
|
|
assert expected_phrase_matches(
|
|
"recognize unknown named entities",
|
|
"Cieľom bolo rozpoznávať neznáme pomenované entity.",
|
|
)
|
|
assert expected_phrase_matches(
|
|
"manually annotated training set",
|
|
"Mal vytvoriť manuálne anotovanú trénovaciu množinu.",
|
|
)
|
|
assert expected_phrase_matches(
|
|
"annotation schema",
|
|
"Mal navrhnúť anotačnú schému.",
|
|
)
|
|
|
|
|
|
def test_mteb_evaluation_can_match_descriptive_answer() -> None:
|
|
assert expected_phrase_matches(
|
|
"Hate Speech a MTEB evaluácia",
|
|
(
|
|
"Stáž zahŕňala slovenský hate speech a hodnotenie "
|
|
"modelov sentence transformer."
|
|
),
|
|
)
|
|
|
|
|
|
def test_ordinary_phrase_still_requires_ordered_local_match() -> None:
|
|
assert not expected_phrase_matches(
|
|
"slovenský internet",
|
|
"Internetový projekt používa slovenský jazyk.",
|
|
)
|
|
|
|
|
|
def test_multi_document_requires_two_gold_sources_not_all() -> None:
|
|
question = {
|
|
"category": "multi_document",
|
|
"expected_answer_contains": ["RAG"],
|
|
"expected_source_urls": [
|
|
"https://example.test/a",
|
|
"https://example.test/b",
|
|
"https://example.test/c",
|
|
"https://example.test/d",
|
|
],
|
|
"should_answer": True,
|
|
}
|
|
answer = (
|
|
"RAG sa nachádza vo viacerých dokumentoch.\n\n"
|
|
"Zdroje:\n"
|
|
"https://example.test/a\n"
|
|
"https://example.test/c"
|
|
)
|
|
|
|
result = evaluate_answer(
|
|
question,
|
|
answer,
|
|
tool_called=True,
|
|
)
|
|
|
|
assert result["source_match_count"] == 2
|
|
assert result["source_required_count"] == 2
|
|
assert result["strict_pass"] is True
|
|
|
|
|
|
def test_invalid_placeholder_source_urls_do_not_force_failure() -> None:
|
|
question = {
|
|
"category": "multi_document",
|
|
"expected_answer_contains": ["strojový preklad"],
|
|
"expected_source_urls": ["...", "...", "..."],
|
|
"should_answer": True,
|
|
}
|
|
|
|
result = evaluate_answer(
|
|
question,
|
|
"Dokumenty riešia strojový preklad.",
|
|
tool_called=True,
|
|
)
|
|
|
|
assert result["source_required_count"] == 0
|
|
assert result["source_url_score"] == 1.0
|
|
assert result["strict_pass"] is True
|
|
|
|
|
|
def test_single_document_still_requires_its_expected_source() -> None:
|
|
question = {
|
|
"category": "specific_fact",
|
|
"expected_answer_contains": ["Prodigy"],
|
|
"expected_source_urls": ["https://example.test/question"],
|
|
"should_answer": True,
|
|
}
|
|
|
|
result = evaluate_answer(
|
|
question,
|
|
"Používa sa Prodigy.",
|
|
tool_called=True,
|
|
)
|
|
|
|
assert result["source_required_count"] == 1
|
|
assert result["strict_pass"] is False
|
|
|
|
def test_cesar_goal_matches_rozoznavat_variant() -> None:
|
|
assert expected_phrase_matches(
|
|
"recognize unknown named entities",
|
|
"Cieľom bolo rozoznávať neznáme pomenované entity.",
|
|
)
|
|
|
|
def test_negative_answer_accepts_semantic_no_answer_wording() -> None:
|
|
question = {
|
|
"category": "unanswerable_missing_fact",
|
|
"expected_answer_contains": [],
|
|
"expected_source_urls": [
|
|
"https://example.test/student"
|
|
],
|
|
"should_answer": False,
|
|
}
|
|
|
|
answer = (
|
|
"V dostupných dokumentoch ZP Wiki sa nepodarilo nájsť "
|
|
"informáciu o konkrétnom modeli GPU."
|
|
)
|
|
|
|
result = evaluate_answer(
|
|
question,
|
|
answer,
|
|
tool_called=True,
|
|
)
|
|
|
|
assert result["returned_no_answer"] is True
|
|
assert result["should_answer_ok"] is True
|
|
assert result["strict_pass"] is True
|
|
|
|
|
|
def test_negative_answer_does_not_require_source_url() -> None:
|
|
question = {
|
|
"category": "unanswerable_missing_fact",
|
|
"expected_answer_contains": [],
|
|
"expected_source_urls": [
|
|
"https://example.test/student"
|
|
],
|
|
"should_answer": False,
|
|
}
|
|
|
|
answer = (
|
|
"V dostupných dokumentoch ZP Wiki sa túto "
|
|
"informáciu nepodarilo spoľahlivo nájsť."
|
|
)
|
|
|
|
result = evaluate_answer(
|
|
question,
|
|
answer,
|
|
tool_called=True,
|
|
)
|
|
|
|
assert result["source_required_count"] == 0
|
|
assert result["source_url_score"] == 1.0
|
|
assert result["strict_pass"] is True
|
|
|
|
def test_spacy_goal_matches_expanded_slovak_wording() -> None:
|
|
assert expected_phrase_matches(
|
|
"Zlepšiť presnosť modelu Spacy pre slovenčinu",
|
|
(
|
|
"Cieľom bolo zlepšiť presnosť modelu Spacy "
|
|
"pre spracovanie slovenčine."
|
|
),
|
|
)
|
|
|
|
|
|
def test_small_context_matches_short_context() -> None:
|
|
assert expected_phrase_matches(
|
|
"príliš maly kontext",
|
|
"Modely mali problém s príliš krátkou dĺžkou kontextu.",
|
|
)
|
|
|
|
|
|
def test_slovak_hate_speech_matches_slovak_translation() -> None:
|
|
assert expected_phrase_matches(
|
|
"Slovak HATE speech",
|
|
"Úloha bola slovenská detekcia nenávistných prejavov.",
|
|
)
|
|
|
|
|
|
def test_slovak_question_answering_matches_translation() -> None:
|
|
assert expected_phrase_matches(
|
|
"Slovak question answering",
|
|
"Išlo o slovenské vyhľadávanie odpovedí na otázky.",
|
|
)
|
|
|
|
|
|
def test_named_matches_menovane_variant() -> None:
|
|
assert expected_phrase_matches(
|
|
"recognize unknown named entities",
|
|
"Cieľom bolo rozpoznať neznáme menované entity.",
|
|
)
|
|
|
|
|
|
def test_multilingual_matches_many_languages_wording() -> None:
|
|
assert expected_phrase_matches(
|
|
"multilinguálna extrakcia trojíc z medicínskych dát",
|
|
(
|
|
"Projekt riešil extrakciu trojíc z lekárskych dát "
|
|
"v mnohých jazykoch."
|
|
),
|
|
)
|
|
|
|
|
|
def test_medical_data_matches_medical_content() -> None:
|
|
assert expected_phrase_matches(
|
|
"Triplet extraction from medical data",
|
|
(
|
|
"Témou bolo multilingválne extrahovanie trojíc "
|
|
"z lekárskeho obsahu."
|
|
),
|
|
)
|