from __future__ import annotations import argparse import csv import json import os import re import statistics import sys import time import unicodedata import urllib.error import urllib.request from collections import defaultdict from datetime import datetime, timezone from pathlib import Path from typing import Any PROJECT_ROOT = Path(__file__).resolve().parents[1] QUESTIONS_PATH = ( PROJECT_ROOT / "evaluation" / "questions.json" ) RESULTS_DIR = ( PROJECT_ROOT / "evaluation" / "results" ) OPENWEBUI_URL = ( "https://ui.tukekemt.xyz/api/chat/completions" ) LOCAL_OPENAPI_URL = ( "http://localhost:8000/openapi.json" ) LOCAL_RAG_URL = ( "http://localhost:8000/rag" ) DEFAULT_MODEL = "model120-fast" DEFAULT_TIMEOUT = 180 NO_ANSWER_TEXT = ( "V dostupných dokumentoch ZP Wiki sa túto " "informáciu nepodarilo spoľahlivo nájsť." ) def parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser( description=( "Answer-level RAG evaluácia cez " "OpenWebUI + Model120 + ZP Agent." ) ) parser.add_argument( "--split", choices=[ "dev", "test", "all", ], default="dev", help="Dataset split. Default: dev.", ) parser.add_argument( "--limit", type=int, default=None, help=( "Maximálny počet otázok. " "Vhodné na kontrolný beh." ), ) parser.add_argument( "--model", default=DEFAULT_MODEL, help=( "OpenWebUI model ID. " f"Default: {DEFAULT_MODEL}." ), ) parser.add_argument( "--questions", type=Path, default=QUESTIONS_PATH, help="Cesta k questions.json.", ) parser.add_argument( "--results-dir", type=Path, default=RESULTS_DIR, help="Adresár pre výsledky.", ) parser.add_argument( "--timeout", type=int, default=DEFAULT_TIMEOUT, help=( "HTTP timeout v sekundách. " f"Default: {DEFAULT_TIMEOUT}." ), ) parser.add_argument( "--delay", type=float, default=0.0, help=( "Pauza medzi otázkami " "v sekundách." ), ) parser.add_argument( "--question-id", action="append", default=[], help=( "Vyhodnoť iba konkrétne ID. " "Možno použiť opakovane." ), ) parser.add_argument( "--fail-fast", action="store_true", help="Pri prvej chybe ukonči evaluáciu.", ) return parser.parse_args() def load_env_value( key: str, env_path: Path | None = None, ) -> str: value = os.environ.get(key) if value: return value if env_path is None: env_path = PROJECT_ROOT / ".env" if not env_path.exists(): raise RuntimeError( f"{key} nie je v environment " f"a {env_path} neexistuje." ) for raw_line in env_path.read_text( encoding="utf-8" ).splitlines(): line = raw_line.strip() if ( not line or line.startswith("#") or "=" not in line ): continue name, value = line.split( "=", 1, ) if name.strip() != key: continue value = value.strip() if ( len(value) >= 2 and value[0] == value[-1] and value[0] in { "'", '"', } ): value = value[1:-1] if value: return value raise RuntimeError( f"{key} sa nepodarilo nájsť." ) def request_json( url: str, *, method: str = "GET", headers: dict[str, str] | None = None, payload: dict[str, Any] | None = None, timeout: int = DEFAULT_TIMEOUT, ) -> dict[str, Any]: data = None if payload is not None: data = json.dumps( payload, ensure_ascii=False, ).encode("utf-8") request = urllib.request.Request( url, data=data, method=method, headers=headers or {}, ) try: with urllib.request.urlopen( request, timeout=timeout, ) as response: raw = response.read().decode( "utf-8" ) except urllib.error.HTTPError as exc: body = exc.read().decode( "utf-8", errors="replace", ) raise RuntimeError( f"HTTP {exc.code} pre {url}: " f"{body[:1500]}" ) from exc except urllib.error.URLError as exc: raise RuntimeError( f"Sieťová chyba pre {url}: " f"{exc}" ) from exc if not raw.strip(): raise RuntimeError( f"Prázdna odpoveď z {url}." ) try: parsed = json.loads(raw) except json.JSONDecodeError as exc: raise RuntimeError( f"Neplatný JSON z {url}: " f"{raw[:1000]}" ) from exc if not isinstance(parsed, dict): raise RuntimeError( f"Očakávaný JSON objekt z {url}, " f"dostal som {type(parsed).__name__}." ) return parsed def resolve_refs( value: Any, document: dict[str, Any], ) -> Any: if isinstance(value, list): return [ resolve_refs( item, document, ) for item in value ] if not isinstance(value, dict): return value ref = value.get("$ref") if ref: if not ref.startswith("#/"): raise RuntimeError( f"Nepodporovaný OpenAPI $ref: {ref}" ) current: Any = document for part in ref[2:].split("/"): current = current[part] resolved = resolve_refs( current, document, ) extra = { key: item for key, item in value.items() if key != "$ref" } if ( extra and isinstance( resolved, dict, ) ): resolved = { **resolved, **resolve_refs( extra, document, ), } return resolved return { key: resolve_refs( item, document, ) for key, item in value.items() } def build_rag_tool( openapi: dict[str, Any], ) -> tuple[str, dict[str, Any]]: try: operation = ( openapi[ "paths" ][ "/rag" ][ "post" ] ) except KeyError as exc: raise RuntimeError( "OpenAPI schéma neobsahuje POST /rag." ) from exc operation_id = operation.get( "operationId" ) if not operation_id: raise RuntimeError( "POST /rag nemá operationId." ) try: schema = ( operation[ "requestBody" ][ "content" ][ "application/json" ][ "schema" ] ) except KeyError as exc: raise RuntimeError( "POST /rag nemá request JSON schema." ) from exc parameters = resolve_refs( schema, openapi, ) description = ( operation.get("description") or operation.get("summary") or ( "Vyhľadá relevantný kontext " "v dokumentoch ZP Wiki." ) ) tool = { "type": "function", "function": { "name": operation_id, "description": description, "parameters": parameters, }, } return operation_id, tool def normalize_text( value: str, ) -> str: value = unicodedata.normalize( "NFKC", value, ) value = value.casefold() value = " ".join( value.split() ) return value MARKDOWN_URL_RE = re.compile( r"\[[^\]]*\]\((https?://[^)]+)\)" ) def normalize_url( value: str, ) -> str: value = value.strip() match = MARKDOWN_URL_RE.search( value ) if match: value = match.group(1) return value.rstrip("/") def extract_urls_from_object( value: Any, ) -> list[str]: result: list[str] = [] if isinstance(value, dict): for key, item in value.items(): if ( key == "source_url" and isinstance( item, str, ) ): result.append( normalize_url(item) ) result.extend( extract_urls_from_object( item ) ) elif isinstance(value, list): for item in value: result.extend( extract_urls_from_object( item ) ) return list( dict.fromkeys(result) ) def get_usage( response: dict[str, Any], ) -> dict[str, int]: usage = response.get( "usage" ) if not isinstance( usage, dict, ): return { "prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0, } return { "prompt_tokens": int( usage.get( "prompt_tokens", 0, ) or 0 ), "completion_tokens": int( usage.get( "completion_tokens", 0, ) or 0 ), "total_tokens": int( usage.get( "total_tokens", 0, ) or 0 ), } def add_usage( total: dict[str, int], current: dict[str, int], ) -> None: for key in ( "prompt_tokens", "completion_tokens", "total_tokens", ): total[key] += current[key] def get_first_message( response: dict[str, Any], ) -> dict[str, Any]: choices = response.get( "choices" ) if ( not isinstance( choices, list, ) or not choices ): raise RuntimeError( "Model nevrátil choices." ) choice = choices[0] if not isinstance( choice, dict, ): raise RuntimeError( "Neplatný choices[0]." ) message = choice.get( "message" ) if not isinstance( message, dict, ): raise RuntimeError( "Model nevrátil message." ) return message def parse_tool_arguments( raw_arguments: Any, ) -> dict[str, Any]: if isinstance( raw_arguments, dict, ): return raw_arguments if not isinstance( raw_arguments, str, ): raise RuntimeError( "Neplatný formát tool arguments." ) try: parsed = json.loads( raw_arguments ) except json.JSONDecodeError as exc: raise RuntimeError( "Model vrátil neplatné JSON " "argumenty toolu: " f"{raw_arguments[:1000]}" ) from exc if not isinstance( parsed, dict, ): raise RuntimeError( "Tool arguments nie sú JSON objekt." ) return parsed def evaluate_answer( question: dict[str, Any], answer: str, *, tool_called: bool, ) -> dict[str, Any]: normalized_answer = normalize_text( answer ) expected_contains = question.get( "expected_answer_contains", [], ) if not isinstance( expected_contains, list, ): expected_contains = [] answer_matches: list[bool] = [] for expected in expected_contains: expected_text = normalize_text( str(expected) ) answer_matches.append( expected_text in normalized_answer ) if answer_matches: answer_contains_score = ( sum(answer_matches) / len(answer_matches) ) else: answer_contains_score = 1.0 expected_urls = question.get( "expected_source_urls", [], ) if not isinstance( expected_urls, list, ): expected_urls = [] normalized_expected_urls = [ normalize_url( str(url) ) for url in expected_urls ] source_matches: list[bool] = [] for expected_url in ( normalized_expected_urls ): source_matches.append( expected_url in answer ) if source_matches: source_url_score = ( sum(source_matches) / len(source_matches) ) else: source_url_score = 1.0 should_answer = bool( question.get( "should_answer", True, ) ) normalized_no_answer = normalize_text( NO_ANSWER_TEXT ) returned_no_answer = ( normalized_no_answer in normalized_answer ) if should_answer: should_answer_ok = ( bool(answer.strip()) and not returned_no_answer ) else: should_answer_ok = ( returned_no_answer ) tool_score = ( 1.0 if tool_called else 0.0 ) should_answer_score = ( 1.0 if should_answer_ok else 0.0 ) overall_score = statistics.mean( [ answer_contains_score, source_url_score, should_answer_score, tool_score, ] ) strict_pass = ( bool(answer.strip()) and all(answer_matches) and all(source_matches) and should_answer_ok and tool_called ) return { "answer_matches": ( answer_matches ), "answer_contains_score": ( answer_contains_score ), "source_matches": ( source_matches ), "source_url_score": ( source_url_score ), "should_answer_ok": ( should_answer_ok ), "should_answer_score": ( should_answer_score ), "returned_no_answer": ( returned_no_answer ), "tool_score": tool_score, "overall_score": overall_score, "strict_pass": strict_pass, } def run_question( question: dict[str, Any], *, model: str, operation_id: str, rag_tool: dict[str, Any], openwebui_api_key: str, search_api_key: str, timeout: int, ) -> dict[str, Any]: started = time.perf_counter() usage_total = { "prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0, } tool_calls_record: list[ dict[str, Any] ] = [] rag_source_urls: list[str] = [] question_text = str( question.get( "question", "", ) ).strip() if not question_text: raise RuntimeError( "Otázka je prázdna." ) messages: list[ dict[str, Any] ] = [ { "role": "user", "content": question_text, } ] first_started = time.perf_counter() first_response = request_json( OPENWEBUI_URL, method="POST", headers={ "Authorization": ( f"Bearer " f"{openwebui_api_key}" ), "Content-Type": ( "application/json" ), }, payload={ "model": model, "messages": messages, "tools": [ rag_tool, ], "tool_choice": "auto", "stream": False, }, timeout=timeout, ) first_latency = ( time.perf_counter() - first_started ) add_usage( usage_total, get_usage( first_response ), ) first_message = get_first_message( first_response ) tool_calls = ( first_message.get( "tool_calls" ) or [] ) if not isinstance( tool_calls, list, ): tool_calls = [] tool_called = bool( tool_calls ) tool_latency_total = 0.0 if not tool_calls: answer = str( first_message.get( "content" ) or "" ) scores = evaluate_answer( question, answer, tool_called=False, ) total_latency = ( time.perf_counter() - started ) return { "answer": answer, "tool_called": False, "tool_call_count": 0, "tool_calls": [], "rag_source_urls": [], "first_model_latency_seconds": ( round( first_latency, 6, ) ), "tool_latency_seconds": 0.0, "final_model_latency_seconds": 0.0, "total_latency_seconds": ( round( total_latency, 6, ) ), "usage": usage_total, "response_model": ( first_response.get( "model" ) ), **scores, } assistant_message = { "role": "assistant", "content": ( first_message.get( "content" ) or "" ), "tool_calls": tool_calls, } messages.append( assistant_message ) for tool_call in tool_calls: if not isinstance( tool_call, dict, ): raise RuntimeError( "Neplatný tool_call objekt." ) function = tool_call.get( "function" ) if not isinstance( function, dict, ): raise RuntimeError( "tool_call nemá function." ) name = function.get( "name" ) if name != operation_id: raise RuntimeError( "Model zavolal neočakávaný " f"tool: {name!r}" ) arguments = parse_tool_arguments( function.get( "arguments" ) ) tool_started = ( time.perf_counter() ) rag_result = request_json( LOCAL_RAG_URL, method="POST", headers={ "X-API-Key": ( search_api_key ), "Content-Type": ( "application/json" ), }, payload=arguments, timeout=timeout, ) tool_latency = ( time.perf_counter() - tool_started ) tool_latency_total += ( tool_latency ) current_urls = ( extract_urls_from_object( rag_result ) ) for url in current_urls: if url not in rag_source_urls: rag_source_urls.append( url ) tool_calls_record.append( { "name": name, "arguments": arguments, "latency_seconds": ( round( tool_latency, 6, ) ), "source_urls": ( current_urls ), } ) tool_call_id = tool_call.get( "id" ) if not tool_call_id: raise RuntimeError( "tool_call nemá id." ) messages.append( { "role": "tool", "tool_call_id": ( tool_call_id ), "name": name, "content": json.dumps( rag_result, ensure_ascii=False, ), } ) final_started = ( time.perf_counter() ) final_response = request_json( OPENWEBUI_URL, method="POST", headers={ "Authorization": ( f"Bearer " f"{openwebui_api_key}" ), "Content-Type": ( "application/json" ), }, payload={ "model": model, "messages": messages, "stream": False, }, timeout=timeout, ) final_latency = ( time.perf_counter() - final_started ) add_usage( usage_total, get_usage( final_response ), ) final_message = get_first_message( final_response ) answer = str( final_message.get( "content" ) or "" ) scores = evaluate_answer( question, answer, tool_called=tool_called, ) total_latency = ( time.perf_counter() - started ) return { "answer": answer, "tool_called": tool_called, "tool_call_count": ( len(tool_calls_record) ), "tool_calls": ( tool_calls_record ), "rag_source_urls": ( rag_source_urls ), "first_model_latency_seconds": ( round( first_latency, 6, ) ), "tool_latency_seconds": ( round( tool_latency_total, 6, ) ), "final_model_latency_seconds": ( round( final_latency, 6, ) ), "total_latency_seconds": ( round( total_latency, 6, ) ), "usage": usage_total, "response_model": ( final_response.get( "model" ) ), **scores, } def load_questions( path: Path, *, split: str, question_ids: list[str], limit: int | None, ) -> list[dict[str, Any]]: data = json.loads( path.read_text( encoding="utf-8" ) ) if not isinstance( data, list, ): raise RuntimeError( "questions.json musí byť JSON pole." ) questions = [ item for item in data if isinstance( item, dict, ) ] if split != "all": questions = [ item for item in questions if item.get("split") == split ] if question_ids: wanted = set( question_ids ) questions = [ item for item in questions if item.get("id") in wanted ] if limit is not None: if limit <= 0: raise RuntimeError( "--limit musí byť > 0." ) questions = questions[ :limit ] return questions def result_prefix( *, split: str, limit: int | None, question_ids: list[str], ) -> str: prefix = ( f"rag_answers_{split}" ) if question_ids: prefix += "_selected" elif limit is not None: prefix += ( f"_limit{limit}" ) return prefix def safe_mean( values: list[float], ) -> float: if not values: return 0.0 return float( statistics.mean(values) ) def build_group_summary( results: list[ dict[str, Any] ], key: str, ) -> dict[str, Any]: groups: dict[ str, list[dict[str, Any]], ] = defaultdict(list) for result in results: group_name = str( result.get( key, "unknown", ) ) groups[ group_name ].append(result) summary: dict[ str, Any, ] = {} for group_name in sorted( groups ): items = groups[ group_name ] summary[ group_name ] = summarize_results( items, include_groups=False, ) return summary def summarize_results( results: list[ dict[str, Any] ], *, include_groups: bool = True, ) -> dict[str, Any]: total = len(results) errors = [ item for item in results if item.get("error") ] completed = total - len( errors ) tool_called_values = [ 1.0 if item.get( "tool_called" ) else 0.0 for item in results ] answer_scores = [ float( item.get( "answer_contains_score", 0.0, ) ) for item in results ] source_scores = [ float( item.get( "source_url_score", 0.0, ) ) for item in results ] should_answer_scores = [ float( item.get( "should_answer_score", 0.0, ) ) for item in results ] overall_scores = [ float( item.get( "overall_score", 0.0, ) ) for item in results ] latencies = [ float( item.get( "total_latency_seconds", 0.0, ) ) for item in results if not item.get( "error" ) ] strict_passes = sum( 1 for item in results if item.get( "strict_pass" ) ) prompt_tokens = sum( int( ( item.get( "usage" ) or {} ).get( "prompt_tokens", 0, ) ) for item in results ) completion_tokens = sum( int( ( item.get( "usage" ) or {} ).get( "completion_tokens", 0, ) ) for item in results ) total_tokens = sum( int( ( item.get( "usage" ) or {} ).get( "total_tokens", 0, ) ) for item in results ) summary = { "total": total, "completed": completed, "errors": len(errors), "tool_call_rate": round( safe_mean( tool_called_values ), 6, ), "answer_contains_score": round( safe_mean( answer_scores ), 6, ), "source_url_score": round( safe_mean( source_scores ), 6, ), "should_answer_score": round( safe_mean( should_answer_scores ), 6, ), "overall_score": round( safe_mean( overall_scores ), 6, ), "strict_pass_count": ( strict_passes ), "strict_pass_rate": round( ( strict_passes / total if total else 0.0 ), 6, ), "mean_latency_seconds": round( safe_mean( latencies ), 6, ), "prompt_tokens": ( prompt_tokens ), "completion_tokens": ( completion_tokens ), "total_tokens": ( total_tokens ), } if include_groups: summary[ "by_category" ] = build_group_summary( results, "category", ) summary[ "by_difficulty" ] = build_group_summary( results, "difficulty", ) return summary def save_json( path: Path, payload: dict[str, Any], ) -> None: path.write_text( json.dumps( payload, ensure_ascii=False, indent=2, ) + "\n", encoding="utf-8", ) def save_csv( path: Path, results: list[ dict[str, Any] ], ) -> None: fieldnames = [ "id", "split", "category", "difficulty", "question", "requested_model", "response_model", "tool_called", "tool_call_count", "answer_contains_score", "source_url_score", "should_answer_score", "overall_score", "strict_pass", "expected_answer_contains", "expected_source_urls", "rag_source_urls", "answer", "first_model_latency_seconds", "tool_latency_seconds", "final_model_latency_seconds", "total_latency_seconds", "prompt_tokens", "completion_tokens", "total_tokens", "error", ] with path.open( "w", encoding="utf-8", newline="", ) as handle: writer = csv.DictWriter( handle, fieldnames=fieldnames, ) writer.writeheader() for result in results: usage = ( result.get( "usage" ) or {} ) writer.writerow( { "id": result.get( "id" ), "split": result.get( "split" ), "category": result.get( "category" ), "difficulty": result.get( "difficulty" ), "question": result.get( "question" ), "requested_model": ( result.get( "requested_model" ) ), "response_model": ( result.get( "response_model" ) ), "tool_called": ( result.get( "tool_called" ) ), "tool_call_count": ( result.get( "tool_call_count" ) ), "answer_contains_score": ( result.get( "answer_contains_score" ) ), "source_url_score": ( result.get( "source_url_score" ) ), "should_answer_score": ( result.get( "should_answer_score" ) ), "overall_score": ( result.get( "overall_score" ) ), "strict_pass": ( result.get( "strict_pass" ) ), "expected_answer_contains": ( json.dumps( result.get( "expected_answer_contains", [], ), ensure_ascii=False, ) ), "expected_source_urls": ( json.dumps( result.get( "expected_source_urls", [], ), ensure_ascii=False, ) ), "rag_source_urls": ( json.dumps( result.get( "rag_source_urls", [], ), ensure_ascii=False, ) ), "answer": result.get( "answer", "", ), "first_model_latency_seconds": ( result.get( "first_model_latency_seconds" ) ), "tool_latency_seconds": ( result.get( "tool_latency_seconds" ) ), "final_model_latency_seconds": ( result.get( "final_model_latency_seconds" ) ), "total_latency_seconds": ( result.get( "total_latency_seconds" ) ), "prompt_tokens": ( usage.get( "prompt_tokens", 0, ) ), "completion_tokens": ( usage.get( "completion_tokens", 0, ) ), "total_tokens": ( usage.get( "total_tokens", 0, ) ), "error": result.get( "error", "", ), } ) def print_summary( summary: dict[str, Any], ) -> None: print() print( "==============================" ) print( "RAG ANSWER EVALUATION" ) print( "==============================" ) print( "Total:", summary["total"], ) print( "Completed:", summary["completed"], ) print( "Errors:", summary["errors"], ) print( "Tool call rate:", f"{summary['tool_call_rate']:.3f}", ) print( "Answer contains:", f"{summary['answer_contains_score']:.3f}", ) print( "Source URL:", f"{summary['source_url_score']:.3f}", ) print( "Should answer:", f"{summary['should_answer_score']:.3f}", ) print( "Overall:", f"{summary['overall_score']:.3f}", ) print( "Strict pass:", ( f"{summary['strict_pass_count']}" f"/{summary['total']} " f"({summary['strict_pass_rate']:.3f})" ), ) print( "Mean latency:", ( f"{summary['mean_latency_seconds']:.3f} s" ), ) print( "Total tokens:", summary["total_tokens"], ) def main() -> int: args = parse_args() questions_path = ( args.questions.resolve() ) if not questions_path.exists(): print( "ERROR: questions.json neexistuje:", questions_path, file=sys.stderr, ) return 2 try: openwebui_api_key = ( load_env_value( "OPENWEBUI_API_KEY" ) ) search_api_key = ( load_env_value( "SEARCH_API_KEY" ) ) except RuntimeError as exc: print( f"ERROR: {exc}", file=sys.stderr, ) return 2 try: questions = load_questions( questions_path, split=args.split, question_ids=( args.question_id ), limit=args.limit, ) except ( RuntimeError, json.JSONDecodeError, ) as exc: print( f"ERROR: {exc}", file=sys.stderr, ) return 2 if not questions: print( "ERROR: Žiadne otázky " "nezodpovedajú filtru.", file=sys.stderr, ) return 2 args.results_dir.mkdir( parents=True, exist_ok=True, ) prefix = result_prefix( split=args.split, limit=args.limit, question_ids=args.question_id, ) json_path = ( args.results_dir / f"{prefix}.json" ) csv_path = ( args.results_dir / f"{prefix}.csv" ) partial_path = ( args.results_dir / f"{prefix}.partial.json" ) print( "Načítavam /openapi.json..." ) try: openapi = request_json( LOCAL_OPENAPI_URL, timeout=args.timeout, ) operation_id, rag_tool = ( build_rag_tool( openapi ) ) except RuntimeError as exc: print( f"ERROR: {exc}", file=sys.stderr, ) return 2 print( "Tool:", operation_id, ) print( "Model:", args.model, ) print( "Split:", args.split, ) print( "Otázky:", len(questions), ) print() results: list[ dict[str, Any] ] = [] total_questions = len( questions ) for index, question in enumerate( questions, start=1, ): question_id = str( question.get( "id", f"unknown-{index}", ) ) question_text = str( question.get( "question", "", ) ) print( f"[{index}/{total_questions}] " f"{question_id}: " f"{question_text}" ) base_result: dict[ str, Any, ] = { "id": question_id, "split": question.get( "split" ), "category": question.get( "category" ), "difficulty": question.get( "difficulty" ), "question": question_text, "requested_model": ( args.model ), "expected_documents": ( question.get( "expected_documents", [], ) ), "expected_source_urls": ( question.get( "expected_source_urls", [], ) ), "expected_answer_contains": ( question.get( "expected_answer_contains", [], ) ), "should_answer": ( question.get( "should_answer", True, ) ), "note": question.get( "note" ), } try: current = run_question( question, model=args.model, operation_id=( operation_id ), rag_tool=rag_tool, openwebui_api_key=( openwebui_api_key ), search_api_key=( search_api_key ), timeout=args.timeout, ) result = { **base_result, **current, "error": None, } print( " tool:", result[ "tool_called" ], "| answer:", ( f"{result['answer_contains_score']:.2f}" ), "| source:", ( f"{result['source_url_score']:.2f}" ), "| strict:", result[ "strict_pass" ], "| latency:", ( f"{result['total_latency_seconds']:.2f}s" ), ) except Exception as exc: result = { **base_result, "answer": "", "tool_called": False, "tool_call_count": 0, "tool_calls": [], "rag_source_urls": [], "first_model_latency_seconds": ( 0.0 ), "tool_latency_seconds": 0.0, "final_model_latency_seconds": ( 0.0 ), "total_latency_seconds": 0.0, "usage": { "prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0, }, "response_model": None, "answer_matches": [], "answer_contains_score": 0.0, "source_matches": [], "source_url_score": 0.0, "should_answer_ok": False, "should_answer_score": 0.0, "returned_no_answer": False, "tool_score": 0.0, "overall_score": 0.0, "strict_pass": False, "error": ( f"{type(exc).__name__}: " f"{exc}" ), } print( " ERROR:", result[ "error" ], ) if args.fail_fast: results.append( result ) break results.append( result ) partial_payload = { "generated_at": ( datetime.now( timezone.utc ).isoformat() ), "status": "partial", "split": args.split, "requested_model": ( args.model ), "questions_path": ( str( questions_path ) ), "completed_so_far": ( len(results) ), "expected_total": ( total_questions ), "summary": ( summarize_results( results ) ), "results": results, } save_json( partial_path, partial_payload, ) if ( args.delay > 0 and index < total_questions ): time.sleep( args.delay ) summary = summarize_results( results ) final_payload = { "generated_at": ( datetime.now( timezone.utc ).isoformat() ), "status": "complete", "split": args.split, "requested_model": ( args.model ), "questions_path": ( str( questions_path ) ), "openwebui_url": ( OPENWEBUI_URL ), "rag_url": ( LOCAL_RAG_URL ), "summary": summary, "results": results, } save_json( json_path, final_payload, ) save_csv( csv_path, results, ) if partial_path.exists(): partial_path.unlink() print_summary( summary ) print() print( "JSON:", json_path, ) print( "CSV:", csv_path, ) if summary["errors"]: return 1 return 0 if __name__ == "__main__": raise SystemExit( main() )