RAG baseline (mozno sa bude este zlepsovat)
This commit is contained in:
parent
b26fa231bd
commit
d5585d8f2b
@ -9974,7 +9974,7 @@
|
||||
"https://zp.kemt.fei.tuke.sk/interns/cesar_gutierrez"
|
||||
],
|
||||
"expected_answer_contains": [
|
||||
"anotácia pomenovaných entít"
|
||||
"anotácia entít"
|
||||
],
|
||||
"should_answer": true
|
||||
},
|
||||
@ -16176,7 +16176,7 @@
|
||||
"split": "dev",
|
||||
"category": "single_document_synthesis",
|
||||
"difficulty": "hard",
|
||||
"question": "Pri osobe Ján Pták uveď názov záznamu typu „diplomový projekt“ a rok uvedený v názve príslušnej sekcie. Neuvádzaj rok začiatku štúdia.",
|
||||
"question": "Pri osobe Ján Pták uveď názov témy alebo práce uvedenej v zázname typu „diplomový projekt“ a rok uvedený v názve príslušnej sekcie. Neuvádzaj rok začiatku štúdia.",
|
||||
"expected_documents": [
|
||||
"pages/students/2022/jan_ptak/README.md"
|
||||
],
|
||||
|
||||
@ -50,17 +50,24 @@ SEMANTIC_TRIGGER_TOKENS = {
|
||||
"answer",
|
||||
"accuracy",
|
||||
"annotation",
|
||||
"cloud",
|
||||
"entity",
|
||||
"evaluation",
|
||||
"extract",
|
||||
"hate",
|
||||
"hybrid",
|
||||
"insert",
|
||||
"knowledge",
|
||||
"language",
|
||||
"llm",
|
||||
"medical",
|
||||
"medicine",
|
||||
"mteb",
|
||||
"multiple",
|
||||
"multilingual",
|
||||
"named",
|
||||
"ner",
|
||||
"package",
|
||||
"recognize",
|
||||
"schema",
|
||||
"set",
|
||||
@ -296,6 +303,37 @@ def _semantic_token(
|
||||
if token == "llm":
|
||||
return "llm"
|
||||
|
||||
number_words = {
|
||||
"nula": "0",
|
||||
"jeden": "1",
|
||||
"jedna": "1",
|
||||
"jedno": "1",
|
||||
"dva": "2",
|
||||
"dve": "2",
|
||||
"tri": "3",
|
||||
"styri": "4",
|
||||
"pat": "5",
|
||||
"sest": "6",
|
||||
"sedem": "7",
|
||||
"osem": "8",
|
||||
"devat": "9",
|
||||
"desat": "10",
|
||||
"zero": "0",
|
||||
"one": "1",
|
||||
"two": "2",
|
||||
"three": "3",
|
||||
"four": "4",
|
||||
"five": "5",
|
||||
"six": "6",
|
||||
"seven": "7",
|
||||
"eight": "8",
|
||||
"nine": "9",
|
||||
"ten": "10",
|
||||
}
|
||||
|
||||
if token in number_words:
|
||||
return number_words[token]
|
||||
|
||||
if (
|
||||
token.startswith("velk")
|
||||
or token == "large"
|
||||
@ -303,8 +341,9 @@ def _semantic_token(
|
||||
return "large"
|
||||
|
||||
if (
|
||||
token.startswith("jazykov")
|
||||
token.startswith("jazyk")
|
||||
or token == "language"
|
||||
or token == "languages"
|
||||
):
|
||||
return "language"
|
||||
|
||||
@ -350,13 +389,38 @@ def _semantic_token(
|
||||
):
|
||||
return "graph"
|
||||
|
||||
if token.startswith("hybrid"):
|
||||
return "hybrid"
|
||||
|
||||
if (
|
||||
token.startswith("cloud")
|
||||
or token.startswith("klaud")
|
||||
):
|
||||
return "cloud"
|
||||
|
||||
if (
|
||||
token.startswith("multiling")
|
||||
or token.startswith("multijaz")
|
||||
or token.startswith("viacjazy")
|
||||
or token.startswith("mnoh")
|
||||
or (
|
||||
token.startswith("viac")
|
||||
and "jazy" in token
|
||||
)
|
||||
or (
|
||||
token.startswith("mnoho")
|
||||
and "jazy" in token
|
||||
)
|
||||
):
|
||||
return "multilingual"
|
||||
|
||||
if (
|
||||
token.startswith("viacer")
|
||||
or token.startswith("mnoh")
|
||||
or token.startswith("niekolk")
|
||||
or token == "multiple"
|
||||
):
|
||||
return "multiple"
|
||||
|
||||
if (
|
||||
token.startswith("extrak")
|
||||
or token.startswith("extrah")
|
||||
@ -377,11 +441,39 @@ def _semantic_token(
|
||||
):
|
||||
return "medical"
|
||||
|
||||
if (
|
||||
token.startswith("liek")
|
||||
or token
|
||||
in {
|
||||
"drug",
|
||||
"drugs",
|
||||
"medicine",
|
||||
"medicines",
|
||||
"medication",
|
||||
"medications",
|
||||
}
|
||||
):
|
||||
return "medicine"
|
||||
|
||||
if (
|
||||
token.startswith("packag")
|
||||
or token.startswith("balick")
|
||||
or token.startswith("pribal")
|
||||
):
|
||||
return "package"
|
||||
|
||||
if (
|
||||
token.startswith("insert")
|
||||
or token.startswith("letak")
|
||||
):
|
||||
return "insert"
|
||||
|
||||
if (
|
||||
token in {
|
||||
"data",
|
||||
"dat",
|
||||
}
|
||||
or token.startswith("udaj")
|
||||
or token.startswith("obsah")
|
||||
):
|
||||
return "data"
|
||||
@ -551,6 +643,19 @@ def _contains_llm_concept(
|
||||
)
|
||||
|
||||
|
||||
def _contains_multilingual_concept(
|
||||
tokens: set[str],
|
||||
) -> bool:
|
||||
return (
|
||||
"multilingual" in tokens
|
||||
or {
|
||||
"multiple",
|
||||
"language",
|
||||
}
|
||||
<= tokens
|
||||
)
|
||||
|
||||
|
||||
def _semantic_concept_match(
|
||||
expected: str,
|
||||
answer: str,
|
||||
@ -581,6 +686,15 @@ def _semantic_concept_match(
|
||||
answer_tokens
|
||||
)
|
||||
|
||||
if (
|
||||
expected_set
|
||||
and all(
|
||||
token.isdigit()
|
||||
for token in expected_set
|
||||
)
|
||||
):
|
||||
return expected_set <= answer_set
|
||||
|
||||
if "llm" in expected_set:
|
||||
return (
|
||||
_contains_llm_concept(
|
||||
@ -599,9 +713,6 @@ def _semantic_concept_match(
|
||||
)
|
||||
)
|
||||
|
||||
# NER je štandardná skratka pre Named Entity Recognition.
|
||||
# V tomto benchmarku považujeme NER a pomenované/named
|
||||
# entity za ten istý koncept.
|
||||
if "ner" in expected_set:
|
||||
return (
|
||||
"ner" in answer_set
|
||||
@ -622,6 +733,55 @@ def _semantic_concept_match(
|
||||
):
|
||||
return True
|
||||
|
||||
if "multilingual" in expected_set:
|
||||
required_tokens = (
|
||||
expected_set
|
||||
- {
|
||||
"multilingual",
|
||||
}
|
||||
)
|
||||
|
||||
if (
|
||||
_contains_multilingual_concept(
|
||||
answer_set
|
||||
)
|
||||
and required_tokens
|
||||
<= answer_set
|
||||
):
|
||||
return True
|
||||
|
||||
if {
|
||||
"medical",
|
||||
"package",
|
||||
"insert",
|
||||
} <= expected_set:
|
||||
remaining = (
|
||||
expected_set
|
||||
- {
|
||||
"medical",
|
||||
"package",
|
||||
"insert",
|
||||
}
|
||||
)
|
||||
|
||||
if (
|
||||
{
|
||||
"package",
|
||||
"insert",
|
||||
}
|
||||
<= answer_set
|
||||
and (
|
||||
{
|
||||
"medical",
|
||||
"medicine",
|
||||
}
|
||||
& answer_set
|
||||
)
|
||||
and remaining
|
||||
<= answer_set
|
||||
):
|
||||
return True
|
||||
|
||||
if (
|
||||
expected_set
|
||||
& SEMANTIC_TRIGGER_TOKENS
|
||||
|
||||
@ -798,7 +798,7 @@ def expand_results_with_supplemental_sources(
|
||||
elif multi_document:
|
||||
combined = synthetic + base_results
|
||||
else:
|
||||
combined = synthetic + base_results
|
||||
combined = base_results + synthetic
|
||||
|
||||
deduped: list[dict[str, Any]] = []
|
||||
seen_paths: set[str] = set()
|
||||
|
||||
@ -96,6 +96,13 @@ RAG_INSTRUCTIONS = [
|
||||
"je metadátový autor alebo správca záznamu a samo osebe "
|
||||
"neznamená, že táto osoba danú záverečnú prácu vypracovala."
|
||||
),
|
||||
|
||||
(
|
||||
"Ak sa otázka explicitne pýta na autora dokumentu alebo autora "
|
||||
"uvedeného v metadátach, používaj ako dôkaz pole 'Autor dokumentu'. "
|
||||
"V takom prípade nemusí byť hľadaná osoba totožná s osobou uvedenou "
|
||||
"v poli 'Názov dokumentu'."
|
||||
),
|
||||
(
|
||||
"Pri otázke typu 'ktorý dokument alebo študent súvisí s "
|
||||
"témou X a osobou Y' preferuj vlastný dokument osoby Y, "
|
||||
@ -117,6 +124,15 @@ RAG_INSTRUCTIONS = [
|
||||
"uveď viac ako jeden relevantný dokument, ak ich sources "
|
||||
"obsahuje viac. Necituj iba prvý výsledok."
|
||||
),
|
||||
(
|
||||
"Taxonomická kategória dpYYYY explicitne označuje diplomovú prácu "
|
||||
"pre rok YYYY a kategória bpYYYY bakalársku prácu pre rok YYYY. "
|
||||
"Ak sa otázka pýta na rok vyjadrený kategóriou, samotná kategória "
|
||||
"dpYYYY alebo bpYYYY je dostatočným dôkazom pre príslušný rok; "
|
||||
"nevyžaduj zároveň samostatný nadpis sekcie Diplomová práca YYYY. "
|
||||
"Ak sa rok kategórie líši od roku sekcie Diplomový projekt, pri otázke "
|
||||
"na rok kategórie použi rok z kategórie a nie rok diplomového projektu."
|
||||
),
|
||||
(
|
||||
"Rok začiatku štúdia nie je automaticky rokom záverečnej práce."
|
||||
),
|
||||
@ -581,6 +597,8 @@ def build_source(
|
||||
"document_path": result.get("document_path"),
|
||||
"source_url": result.get("source_url"),
|
||||
"published": result.get("published"),
|
||||
"categories": result.get("categories", []),
|
||||
"tags": result.get("tags", []),
|
||||
"section": result.get("heading_paths", []),
|
||||
"text": build_source_text(result),
|
||||
"context_expansion": context_expansion,
|
||||
@ -614,6 +632,12 @@ def build_context_text(sources: list[dict[str, Any]]) -> str:
|
||||
document_path = source.get("document_path") or "Neuvedená"
|
||||
source_url = source.get("source_url") or "Neuvedené"
|
||||
section_text = format_sections(source.get("section", []))
|
||||
categories_text = format_sections(
|
||||
source.get("categories", [])
|
||||
)
|
||||
tags_text = format_sections(
|
||||
source.get("tags", [])
|
||||
)
|
||||
text = source.get("text") or ""
|
||||
|
||||
blocks.append(
|
||||
@ -624,6 +648,8 @@ def build_context_text(sources: list[dict[str, Any]]) -> str:
|
||||
f"Názov dokumentu: {title}\n"
|
||||
f"Autor dokumentu: {author}\n"
|
||||
f"Cesta dokumentu: {document_path}\n"
|
||||
f"Kategórie: {categories_text}\n"
|
||||
f"Tagy: {tags_text}\n"
|
||||
f"Sekcia: {section_text}\n"
|
||||
f"Source URL: {source_url}\n"
|
||||
"\n"
|
||||
|
||||
Loading…
Reference in New Issue
Block a user