from __future__ import annotations from evaluation.rag_metrics import ( evaluate_answer, expected_phrase_matches, ) def test_compact_unit_spacing_matches() -> None: assert expected_phrase_matches( "450MB", "Nazbieraných bolo približne 450 MB textu.", ) def test_llm_matches_large_language_model() -> None: assert expected_phrase_matches( "LLM", "Projekt používa veľký jazykový model.", ) def test_named_entities_matches_slovak_wording() -> None: assert expected_phrase_matches( "named entities", "Stáž sa venovala pomenovaným entitám.", ) def test_knowledge_graph_matches_slovak_wording() -> None: assert expected_phrase_matches( "knowledge graph", "Cieľom bolo zostaviť znalostný graf.", ) def test_multilingual_medical_triplet_phrase_matches_translation() -> None: assert expected_phrase_matches( "multilinguálna extrakcia trojíc z medicínskych dát", "Téma bola viacjazyčné extrahovanie trojíc z lekárskych dát.", ) def test_cesar_goal_matches_slovak_equivalent() -> None: assert expected_phrase_matches( "recognize unknown named entities", "Cieľom bolo rozpoznávať neznáme pomenované entity.", ) assert expected_phrase_matches( "manually annotated training set", "Mal vytvoriť manuálne anotovanú trénovaciu množinu.", ) assert expected_phrase_matches( "annotation schema", "Mal navrhnúť anotačnú schému.", ) def test_mteb_evaluation_can_match_descriptive_answer() -> None: assert expected_phrase_matches( "Hate Speech a MTEB evaluácia", ( "Stáž zahŕňala slovenský hate speech a hodnotenie " "modelov sentence transformer." ), ) def test_ordinary_phrase_still_requires_ordered_local_match() -> None: assert not expected_phrase_matches( "slovenský internet", "Internetový projekt používa slovenský jazyk.", ) def test_multi_document_requires_two_gold_sources_not_all() -> None: question = { "category": "multi_document", "expected_answer_contains": ["RAG"], "expected_source_urls": [ "https://example.test/a", "https://example.test/b", "https://example.test/c", "https://example.test/d", ], "should_answer": True, } answer = ( "RAG sa nachádza vo viacerých dokumentoch.\n\n" "Zdroje:\n" "https://example.test/a\n" "https://example.test/c" ) result = evaluate_answer( question, answer, tool_called=True, ) assert result["source_match_count"] == 2 assert result["source_required_count"] == 2 assert result["strict_pass"] is True def test_invalid_placeholder_source_urls_do_not_force_failure() -> None: question = { "category": "multi_document", "expected_answer_contains": ["strojový preklad"], "expected_source_urls": ["...", "...", "..."], "should_answer": True, } result = evaluate_answer( question, "Dokumenty riešia strojový preklad.", tool_called=True, ) assert result["source_required_count"] == 0 assert result["source_url_score"] == 1.0 assert result["strict_pass"] is True def test_single_document_still_requires_its_expected_source() -> None: question = { "category": "specific_fact", "expected_answer_contains": ["Prodigy"], "expected_source_urls": ["https://example.test/question"], "should_answer": True, } result = evaluate_answer( question, "Používa sa Prodigy.", tool_called=True, ) assert result["source_required_count"] == 1 assert result["strict_pass"] is False def test_cesar_goal_matches_rozoznavat_variant() -> None: assert expected_phrase_matches( "recognize unknown named entities", "Cieľom bolo rozoznávať neznáme pomenované entity.", )