diff --git a/test/test_rag_metrics_semantic.py b/test/test_rag_metrics_semantic.py new file mode 100644 index 0000000..6b501c3 --- /dev/null +++ b/test/test_rag_metrics_semantic.py @@ -0,0 +1,146 @@ +from __future__ import annotations + +from evaluation.rag_metrics import ( + evaluate_answer, + expected_phrase_matches, +) + + +def test_compact_unit_spacing_matches() -> None: + assert expected_phrase_matches( + "450MB", + "Nazbieraných bolo približne 450 MB textu.", + ) + + +def test_llm_matches_large_language_model() -> None: + assert expected_phrase_matches( + "LLM", + "Projekt používa veľký jazykový model.", + ) + + +def test_named_entities_matches_slovak_wording() -> None: + assert expected_phrase_matches( + "named entities", + "Stáž sa venovala pomenovaným entitám.", + ) + + +def test_knowledge_graph_matches_slovak_wording() -> None: + assert expected_phrase_matches( + "knowledge graph", + "Cieľom bolo zostaviť znalostný graf.", + ) + + +def test_multilingual_medical_triplet_phrase_matches_translation() -> None: + assert expected_phrase_matches( + "multilinguálna extrakcia trojíc z medicínskych dát", + "Téma bola viacjazyčné extrahovanie trojíc z lekárskych dát.", + ) + + +def test_cesar_goal_matches_slovak_equivalent() -> None: + assert expected_phrase_matches( + "recognize unknown named entities", + "Cieľom bolo rozpoznávať neznáme pomenované entity.", + ) + assert expected_phrase_matches( + "manually annotated training set", + "Mal vytvoriť manuálne anotovanú trénovaciu množinu.", + ) + assert expected_phrase_matches( + "annotation schema", + "Mal navrhnúť anotačnú schému.", + ) + + +def test_mteb_evaluation_can_match_descriptive_answer() -> None: + assert expected_phrase_matches( + "Hate Speech a MTEB evaluácia", + ( + "Stáž zahŕňala slovenský hate speech a hodnotenie " + "modelov sentence transformer." + ), + ) + + +def test_ordinary_phrase_still_requires_ordered_local_match() -> None: + assert not expected_phrase_matches( + "slovenský internet", + "Internetový projekt používa slovenský jazyk.", + ) + + +def test_multi_document_requires_two_gold_sources_not_all() -> None: + question = { + "category": "multi_document", + "expected_answer_contains": ["RAG"], + "expected_source_urls": [ + "https://example.test/a", + "https://example.test/b", + "https://example.test/c", + "https://example.test/d", + ], + "should_answer": True, + } + answer = ( + "RAG sa nachádza vo viacerých dokumentoch.\n\n" + "Zdroje:\n" + "https://example.test/a\n" + "https://example.test/c" + ) + + result = evaluate_answer( + question, + answer, + tool_called=True, + ) + + assert result["source_match_count"] == 2 + assert result["source_required_count"] == 2 + assert result["strict_pass"] is True + + +def test_invalid_placeholder_source_urls_do_not_force_failure() -> None: + question = { + "category": "multi_document", + "expected_answer_contains": ["strojový preklad"], + "expected_source_urls": ["...", "...", "..."], + "should_answer": True, + } + + result = evaluate_answer( + question, + "Dokumenty riešia strojový preklad.", + tool_called=True, + ) + + assert result["source_required_count"] == 0 + assert result["source_url_score"] == 1.0 + assert result["strict_pass"] is True + + +def test_single_document_still_requires_its_expected_source() -> None: + question = { + "category": "specific_fact", + "expected_answer_contains": ["Prodigy"], + "expected_source_urls": ["https://example.test/question"], + "should_answer": True, + } + + result = evaluate_answer( + question, + "Používa sa Prodigy.", + tool_called=True, + ) + + assert result["source_required_count"] == 1 + assert result["strict_pass"] is False + +def test_cesar_goal_matches_rozoznavat_variant() -> None: + assert expected_phrase_matches( + "recognize unknown named entities", + "Cieľom bolo rozoznávať neznáme pomenované entity.", + )