dp-zp-agent/test/test_rag_metrics_semantic.py

147 lines
4.0 KiB
Python

from __future__ import annotations
from evaluation.rag_metrics import (
evaluate_answer,
expected_phrase_matches,
)
def test_compact_unit_spacing_matches() -> None:
assert expected_phrase_matches(
"450MB",
"Nazbieraných bolo približne 450 MB textu.",
)
def test_llm_matches_large_language_model() -> None:
assert expected_phrase_matches(
"LLM",
"Projekt používa veľký jazykový model.",
)
def test_named_entities_matches_slovak_wording() -> None:
assert expected_phrase_matches(
"named entities",
"Stáž sa venovala pomenovaným entitám.",
)
def test_knowledge_graph_matches_slovak_wording() -> None:
assert expected_phrase_matches(
"knowledge graph",
"Cieľom bolo zostaviť znalostný graf.",
)
def test_multilingual_medical_triplet_phrase_matches_translation() -> None:
assert expected_phrase_matches(
"multilinguálna extrakcia trojíc z medicínskych dát",
"Téma bola viacjazyčné extrahovanie trojíc z lekárskych dát.",
)
def test_cesar_goal_matches_slovak_equivalent() -> None:
assert expected_phrase_matches(
"recognize unknown named entities",
"Cieľom bolo rozpoznávať neznáme pomenované entity.",
)
assert expected_phrase_matches(
"manually annotated training set",
"Mal vytvoriť manuálne anotovanú trénovaciu množinu.",
)
assert expected_phrase_matches(
"annotation schema",
"Mal navrhnúť anotačnú schému.",
)
def test_mteb_evaluation_can_match_descriptive_answer() -> None:
assert expected_phrase_matches(
"Hate Speech a MTEB evaluácia",
(
"Stáž zahŕňala slovenský hate speech a hodnotenie "
"modelov sentence transformer."
),
)
def test_ordinary_phrase_still_requires_ordered_local_match() -> None:
assert not expected_phrase_matches(
"slovenský internet",
"Internetový projekt používa slovenský jazyk.",
)
def test_multi_document_requires_two_gold_sources_not_all() -> None:
question = {
"category": "multi_document",
"expected_answer_contains": ["RAG"],
"expected_source_urls": [
"https://example.test/a",
"https://example.test/b",
"https://example.test/c",
"https://example.test/d",
],
"should_answer": True,
}
answer = (
"RAG sa nachádza vo viacerých dokumentoch.\n\n"
"Zdroje:\n"
"https://example.test/a\n"
"https://example.test/c"
)
result = evaluate_answer(
question,
answer,
tool_called=True,
)
assert result["source_match_count"] == 2
assert result["source_required_count"] == 2
assert result["strict_pass"] is True
def test_invalid_placeholder_source_urls_do_not_force_failure() -> None:
question = {
"category": "multi_document",
"expected_answer_contains": ["strojový preklad"],
"expected_source_urls": ["...", "...", "..."],
"should_answer": True,
}
result = evaluate_answer(
question,
"Dokumenty riešia strojový preklad.",
tool_called=True,
)
assert result["source_required_count"] == 0
assert result["source_url_score"] == 1.0
assert result["strict_pass"] is True
def test_single_document_still_requires_its_expected_source() -> None:
question = {
"category": "specific_fact",
"expected_answer_contains": ["Prodigy"],
"expected_source_urls": ["https://example.test/question"],
"should_answer": True,
}
result = evaluate_answer(
question,
"Používa sa Prodigy.",
tool_called=True,
)
assert result["source_required_count"] == 1
assert result["strict_pass"] is False
def test_cesar_goal_matches_rozoznavat_variant() -> None:
assert expected_phrase_matches(
"recognize unknown named entities",
"Cieľom bolo rozoznávať neznáme pomenované entity.",
)