RAG baseline (mozno sa bude este zlepsovat)
This commit is contained in:
parent
b26fa231bd
commit
d5585d8f2b
@ -9974,7 +9974,7 @@
|
|||||||
"https://zp.kemt.fei.tuke.sk/interns/cesar_gutierrez"
|
"https://zp.kemt.fei.tuke.sk/interns/cesar_gutierrez"
|
||||||
],
|
],
|
||||||
"expected_answer_contains": [
|
"expected_answer_contains": [
|
||||||
"anotácia pomenovaných entít"
|
"anotácia entít"
|
||||||
],
|
],
|
||||||
"should_answer": true
|
"should_answer": true
|
||||||
},
|
},
|
||||||
@ -16176,7 +16176,7 @@
|
|||||||
"split": "dev",
|
"split": "dev",
|
||||||
"category": "single_document_synthesis",
|
"category": "single_document_synthesis",
|
||||||
"difficulty": "hard",
|
"difficulty": "hard",
|
||||||
"question": "Pri osobe Ján Pták uveď názov záznamu typu „diplomový projekt“ a rok uvedený v názve príslušnej sekcie. Neuvádzaj rok začiatku štúdia.",
|
"question": "Pri osobe Ján Pták uveď názov témy alebo práce uvedenej v zázname typu „diplomový projekt“ a rok uvedený v názve príslušnej sekcie. Neuvádzaj rok začiatku štúdia.",
|
||||||
"expected_documents": [
|
"expected_documents": [
|
||||||
"pages/students/2022/jan_ptak/README.md"
|
"pages/students/2022/jan_ptak/README.md"
|
||||||
],
|
],
|
||||||
|
|||||||
@ -50,17 +50,24 @@ SEMANTIC_TRIGGER_TOKENS = {
|
|||||||
"answer",
|
"answer",
|
||||||
"accuracy",
|
"accuracy",
|
||||||
"annotation",
|
"annotation",
|
||||||
|
"cloud",
|
||||||
"entity",
|
"entity",
|
||||||
"evaluation",
|
"evaluation",
|
||||||
"extract",
|
"extract",
|
||||||
"hate",
|
"hate",
|
||||||
|
"hybrid",
|
||||||
|
"insert",
|
||||||
"knowledge",
|
"knowledge",
|
||||||
|
"language",
|
||||||
"llm",
|
"llm",
|
||||||
"medical",
|
"medical",
|
||||||
|
"medicine",
|
||||||
"mteb",
|
"mteb",
|
||||||
|
"multiple",
|
||||||
"multilingual",
|
"multilingual",
|
||||||
"named",
|
"named",
|
||||||
"ner",
|
"ner",
|
||||||
|
"package",
|
||||||
"recognize",
|
"recognize",
|
||||||
"schema",
|
"schema",
|
||||||
"set",
|
"set",
|
||||||
@ -296,6 +303,37 @@ def _semantic_token(
|
|||||||
if token == "llm":
|
if token == "llm":
|
||||||
return "llm"
|
return "llm"
|
||||||
|
|
||||||
|
number_words = {
|
||||||
|
"nula": "0",
|
||||||
|
"jeden": "1",
|
||||||
|
"jedna": "1",
|
||||||
|
"jedno": "1",
|
||||||
|
"dva": "2",
|
||||||
|
"dve": "2",
|
||||||
|
"tri": "3",
|
||||||
|
"styri": "4",
|
||||||
|
"pat": "5",
|
||||||
|
"sest": "6",
|
||||||
|
"sedem": "7",
|
||||||
|
"osem": "8",
|
||||||
|
"devat": "9",
|
||||||
|
"desat": "10",
|
||||||
|
"zero": "0",
|
||||||
|
"one": "1",
|
||||||
|
"two": "2",
|
||||||
|
"three": "3",
|
||||||
|
"four": "4",
|
||||||
|
"five": "5",
|
||||||
|
"six": "6",
|
||||||
|
"seven": "7",
|
||||||
|
"eight": "8",
|
||||||
|
"nine": "9",
|
||||||
|
"ten": "10",
|
||||||
|
}
|
||||||
|
|
||||||
|
if token in number_words:
|
||||||
|
return number_words[token]
|
||||||
|
|
||||||
if (
|
if (
|
||||||
token.startswith("velk")
|
token.startswith("velk")
|
||||||
or token == "large"
|
or token == "large"
|
||||||
@ -303,8 +341,9 @@ def _semantic_token(
|
|||||||
return "large"
|
return "large"
|
||||||
|
|
||||||
if (
|
if (
|
||||||
token.startswith("jazykov")
|
token.startswith("jazyk")
|
||||||
or token == "language"
|
or token == "language"
|
||||||
|
or token == "languages"
|
||||||
):
|
):
|
||||||
return "language"
|
return "language"
|
||||||
|
|
||||||
@ -350,13 +389,38 @@ def _semantic_token(
|
|||||||
):
|
):
|
||||||
return "graph"
|
return "graph"
|
||||||
|
|
||||||
|
if token.startswith("hybrid"):
|
||||||
|
return "hybrid"
|
||||||
|
|
||||||
|
if (
|
||||||
|
token.startswith("cloud")
|
||||||
|
or token.startswith("klaud")
|
||||||
|
):
|
||||||
|
return "cloud"
|
||||||
|
|
||||||
if (
|
if (
|
||||||
token.startswith("multiling")
|
token.startswith("multiling")
|
||||||
|
or token.startswith("multijaz")
|
||||||
or token.startswith("viacjazy")
|
or token.startswith("viacjazy")
|
||||||
or token.startswith("mnoh")
|
or (
|
||||||
|
token.startswith("viac")
|
||||||
|
and "jazy" in token
|
||||||
|
)
|
||||||
|
or (
|
||||||
|
token.startswith("mnoho")
|
||||||
|
and "jazy" in token
|
||||||
|
)
|
||||||
):
|
):
|
||||||
return "multilingual"
|
return "multilingual"
|
||||||
|
|
||||||
|
if (
|
||||||
|
token.startswith("viacer")
|
||||||
|
or token.startswith("mnoh")
|
||||||
|
or token.startswith("niekolk")
|
||||||
|
or token == "multiple"
|
||||||
|
):
|
||||||
|
return "multiple"
|
||||||
|
|
||||||
if (
|
if (
|
||||||
token.startswith("extrak")
|
token.startswith("extrak")
|
||||||
or token.startswith("extrah")
|
or token.startswith("extrah")
|
||||||
@ -377,11 +441,39 @@ def _semantic_token(
|
|||||||
):
|
):
|
||||||
return "medical"
|
return "medical"
|
||||||
|
|
||||||
|
if (
|
||||||
|
token.startswith("liek")
|
||||||
|
or token
|
||||||
|
in {
|
||||||
|
"drug",
|
||||||
|
"drugs",
|
||||||
|
"medicine",
|
||||||
|
"medicines",
|
||||||
|
"medication",
|
||||||
|
"medications",
|
||||||
|
}
|
||||||
|
):
|
||||||
|
return "medicine"
|
||||||
|
|
||||||
|
if (
|
||||||
|
token.startswith("packag")
|
||||||
|
or token.startswith("balick")
|
||||||
|
or token.startswith("pribal")
|
||||||
|
):
|
||||||
|
return "package"
|
||||||
|
|
||||||
|
if (
|
||||||
|
token.startswith("insert")
|
||||||
|
or token.startswith("letak")
|
||||||
|
):
|
||||||
|
return "insert"
|
||||||
|
|
||||||
if (
|
if (
|
||||||
token in {
|
token in {
|
||||||
"data",
|
"data",
|
||||||
"dat",
|
"dat",
|
||||||
}
|
}
|
||||||
|
or token.startswith("udaj")
|
||||||
or token.startswith("obsah")
|
or token.startswith("obsah")
|
||||||
):
|
):
|
||||||
return "data"
|
return "data"
|
||||||
@ -551,6 +643,19 @@ def _contains_llm_concept(
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _contains_multilingual_concept(
|
||||||
|
tokens: set[str],
|
||||||
|
) -> bool:
|
||||||
|
return (
|
||||||
|
"multilingual" in tokens
|
||||||
|
or {
|
||||||
|
"multiple",
|
||||||
|
"language",
|
||||||
|
}
|
||||||
|
<= tokens
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def _semantic_concept_match(
|
def _semantic_concept_match(
|
||||||
expected: str,
|
expected: str,
|
||||||
answer: str,
|
answer: str,
|
||||||
@ -581,6 +686,15 @@ def _semantic_concept_match(
|
|||||||
answer_tokens
|
answer_tokens
|
||||||
)
|
)
|
||||||
|
|
||||||
|
if (
|
||||||
|
expected_set
|
||||||
|
and all(
|
||||||
|
token.isdigit()
|
||||||
|
for token in expected_set
|
||||||
|
)
|
||||||
|
):
|
||||||
|
return expected_set <= answer_set
|
||||||
|
|
||||||
if "llm" in expected_set:
|
if "llm" in expected_set:
|
||||||
return (
|
return (
|
||||||
_contains_llm_concept(
|
_contains_llm_concept(
|
||||||
@ -599,9 +713,6 @@ def _semantic_concept_match(
|
|||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
|
||||||
# NER je štandardná skratka pre Named Entity Recognition.
|
|
||||||
# V tomto benchmarku považujeme NER a pomenované/named
|
|
||||||
# entity za ten istý koncept.
|
|
||||||
if "ner" in expected_set:
|
if "ner" in expected_set:
|
||||||
return (
|
return (
|
||||||
"ner" in answer_set
|
"ner" in answer_set
|
||||||
@ -622,6 +733,55 @@ def _semantic_concept_match(
|
|||||||
):
|
):
|
||||||
return True
|
return True
|
||||||
|
|
||||||
|
if "multilingual" in expected_set:
|
||||||
|
required_tokens = (
|
||||||
|
expected_set
|
||||||
|
- {
|
||||||
|
"multilingual",
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
if (
|
||||||
|
_contains_multilingual_concept(
|
||||||
|
answer_set
|
||||||
|
)
|
||||||
|
and required_tokens
|
||||||
|
<= answer_set
|
||||||
|
):
|
||||||
|
return True
|
||||||
|
|
||||||
|
if {
|
||||||
|
"medical",
|
||||||
|
"package",
|
||||||
|
"insert",
|
||||||
|
} <= expected_set:
|
||||||
|
remaining = (
|
||||||
|
expected_set
|
||||||
|
- {
|
||||||
|
"medical",
|
||||||
|
"package",
|
||||||
|
"insert",
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
if (
|
||||||
|
{
|
||||||
|
"package",
|
||||||
|
"insert",
|
||||||
|
}
|
||||||
|
<= answer_set
|
||||||
|
and (
|
||||||
|
{
|
||||||
|
"medical",
|
||||||
|
"medicine",
|
||||||
|
}
|
||||||
|
& answer_set
|
||||||
|
)
|
||||||
|
and remaining
|
||||||
|
<= answer_set
|
||||||
|
):
|
||||||
|
return True
|
||||||
|
|
||||||
if (
|
if (
|
||||||
expected_set
|
expected_set
|
||||||
& SEMANTIC_TRIGGER_TOKENS
|
& SEMANTIC_TRIGGER_TOKENS
|
||||||
|
|||||||
@ -798,7 +798,7 @@ def expand_results_with_supplemental_sources(
|
|||||||
elif multi_document:
|
elif multi_document:
|
||||||
combined = synthetic + base_results
|
combined = synthetic + base_results
|
||||||
else:
|
else:
|
||||||
combined = synthetic + base_results
|
combined = base_results + synthetic
|
||||||
|
|
||||||
deduped: list[dict[str, Any]] = []
|
deduped: list[dict[str, Any]] = []
|
||||||
seen_paths: set[str] = set()
|
seen_paths: set[str] = set()
|
||||||
|
|||||||
@ -96,6 +96,13 @@ RAG_INSTRUCTIONS = [
|
|||||||
"je metadátový autor alebo správca záznamu a samo osebe "
|
"je metadátový autor alebo správca záznamu a samo osebe "
|
||||||
"neznamená, že táto osoba danú záverečnú prácu vypracovala."
|
"neznamená, že táto osoba danú záverečnú prácu vypracovala."
|
||||||
),
|
),
|
||||||
|
|
||||||
|
(
|
||||||
|
"Ak sa otázka explicitne pýta na autora dokumentu alebo autora "
|
||||||
|
"uvedeného v metadátach, používaj ako dôkaz pole 'Autor dokumentu'. "
|
||||||
|
"V takom prípade nemusí byť hľadaná osoba totožná s osobou uvedenou "
|
||||||
|
"v poli 'Názov dokumentu'."
|
||||||
|
),
|
||||||
(
|
(
|
||||||
"Pri otázke typu 'ktorý dokument alebo študent súvisí s "
|
"Pri otázke typu 'ktorý dokument alebo študent súvisí s "
|
||||||
"témou X a osobou Y' preferuj vlastný dokument osoby Y, "
|
"témou X a osobou Y' preferuj vlastný dokument osoby Y, "
|
||||||
@ -117,6 +124,15 @@ RAG_INSTRUCTIONS = [
|
|||||||
"uveď viac ako jeden relevantný dokument, ak ich sources "
|
"uveď viac ako jeden relevantný dokument, ak ich sources "
|
||||||
"obsahuje viac. Necituj iba prvý výsledok."
|
"obsahuje viac. Necituj iba prvý výsledok."
|
||||||
),
|
),
|
||||||
|
(
|
||||||
|
"Taxonomická kategória dpYYYY explicitne označuje diplomovú prácu "
|
||||||
|
"pre rok YYYY a kategória bpYYYY bakalársku prácu pre rok YYYY. "
|
||||||
|
"Ak sa otázka pýta na rok vyjadrený kategóriou, samotná kategória "
|
||||||
|
"dpYYYY alebo bpYYYY je dostatočným dôkazom pre príslušný rok; "
|
||||||
|
"nevyžaduj zároveň samostatný nadpis sekcie Diplomová práca YYYY. "
|
||||||
|
"Ak sa rok kategórie líši od roku sekcie Diplomový projekt, pri otázke "
|
||||||
|
"na rok kategórie použi rok z kategórie a nie rok diplomového projektu."
|
||||||
|
),
|
||||||
(
|
(
|
||||||
"Rok začiatku štúdia nie je automaticky rokom záverečnej práce."
|
"Rok začiatku štúdia nie je automaticky rokom záverečnej práce."
|
||||||
),
|
),
|
||||||
@ -581,6 +597,8 @@ def build_source(
|
|||||||
"document_path": result.get("document_path"),
|
"document_path": result.get("document_path"),
|
||||||
"source_url": result.get("source_url"),
|
"source_url": result.get("source_url"),
|
||||||
"published": result.get("published"),
|
"published": result.get("published"),
|
||||||
|
"categories": result.get("categories", []),
|
||||||
|
"tags": result.get("tags", []),
|
||||||
"section": result.get("heading_paths", []),
|
"section": result.get("heading_paths", []),
|
||||||
"text": build_source_text(result),
|
"text": build_source_text(result),
|
||||||
"context_expansion": context_expansion,
|
"context_expansion": context_expansion,
|
||||||
@ -614,6 +632,12 @@ def build_context_text(sources: list[dict[str, Any]]) -> str:
|
|||||||
document_path = source.get("document_path") or "Neuvedená"
|
document_path = source.get("document_path") or "Neuvedená"
|
||||||
source_url = source.get("source_url") or "Neuvedené"
|
source_url = source.get("source_url") or "Neuvedené"
|
||||||
section_text = format_sections(source.get("section", []))
|
section_text = format_sections(source.get("section", []))
|
||||||
|
categories_text = format_sections(
|
||||||
|
source.get("categories", [])
|
||||||
|
)
|
||||||
|
tags_text = format_sections(
|
||||||
|
source.get("tags", [])
|
||||||
|
)
|
||||||
text = source.get("text") or ""
|
text = source.get("text") or ""
|
||||||
|
|
||||||
blocks.append(
|
blocks.append(
|
||||||
@ -624,6 +648,8 @@ def build_context_text(sources: list[dict[str, Any]]) -> str:
|
|||||||
f"Názov dokumentu: {title}\n"
|
f"Názov dokumentu: {title}\n"
|
||||||
f"Autor dokumentu: {author}\n"
|
f"Autor dokumentu: {author}\n"
|
||||||
f"Cesta dokumentu: {document_path}\n"
|
f"Cesta dokumentu: {document_path}\n"
|
||||||
|
f"Kategórie: {categories_text}\n"
|
||||||
|
f"Tagy: {tags_text}\n"
|
||||||
f"Sekcia: {section_text}\n"
|
f"Sekcia: {section_text}\n"
|
||||||
f"Source URL: {source_url}\n"
|
f"Source URL: {source_url}\n"
|
||||||
"\n"
|
"\n"
|
||||||
|
|||||||
Loading…
Reference in New Issue
Block a user