from __future__ import annotations import json import sqlite3 from pathlib import Path import pytest import scripts.build_sqlite_index as indexer def write_json(path: Path, value) -> None: path.parent.mkdir( parents=True, exist_ok=True, ) path.write_text( json.dumps( value, ensure_ascii=False, ), encoding="utf-8", ) def sample_documents() -> list[dict]: return [ { "path": "pages/test/README.md", "title": "Strojový preklad", "author": "Autor", "published": True, "content_length": 100, "metadata": { "published": True, }, } ] def sample_chunks() -> list[dict]: return [ { "chunk_id": ( "pages/test/README.md::chunk-0" ), "document_path": ( "pages/test/README.md" ), "title": "Strojový preklad", "author": "Autor", "published": True, "chunk_index": 0, "heading_paths": [ ["Úvod"] ], "text": ( "Dokument: Strojový preklad. " "Neurónový preklad textu." ), "text_length": 58, "token_count": 16, "content_hash": "abc", "tags": [ "translation", "nlp", ], "categories": [ "project", ], } ] def configure_indexer( monkeypatch: pytest.MonkeyPatch, tmp_path: Path, ) -> tuple[ Path, Path, Path, ]: documents_file = ( tmp_path / "documents.json" ) chunks_file = ( tmp_path / "chunks.json" ) db_file = ( tmp_path / "zp_index.sqlite" ) write_json( documents_file, sample_documents(), ) write_json( chunks_file, sample_chunks(), ) monkeypatch.setattr( indexer, "DOCUMENTS_FILE", documents_file, ) monkeypatch.setattr( indexer, "CHUNKS_FILE", chunks_file, ) monkeypatch.setattr( indexer, "DB_FILE", db_file, ) return ( documents_file, chunks_file, db_file, ) def test_database_contains_documents_chunks_metadata_and_fts( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: _, _, db_file = configure_indexer( monkeypatch, tmp_path, ) counts = indexer.build_database() assert counts == { "documents": 1, "chunks": 1, "fts_chunks": 1, "embedding_chunks": 1, "tags": 2, "categories": 1, } with sqlite3.connect( db_file ) as conn: assert ( conn.execute( "PRAGMA integrity_check" ).fetchone()[0] == "ok" ) assert ( conn.execute( "PRAGMA foreign_key_check" ).fetchall() == [] ) assert ( conn.execute( """ SELECT published FROM chunks """ ).fetchone()[0] == 1 ) assert ( conn.execute( """ SELECT COUNT(*) FROM chunk_tags """ ).fetchone()[0] == 2 ) assert ( conn.execute( """ SELECT COUNT(*) FROM chunk_categories """ ).fetchone()[0] == 1 ) assert ( conn.execute( """ SELECT COUNT(*) FROM chunks_fts WHERE chunks_fts MATCH 'strojovy' """ ).fetchone()[0] == 1 ) assert ( conn.execute( """ SELECT COUNT(*) FROM chunk_embeddings """ ).fetchone()[0] == 1 ) embedding_row = conn.execute( """ SELECT model, dimensions, LENGTH(embedding) FROM chunk_embeddings """ ).fetchone() assert embedding_row is not None model, dimensions, blob_size = ( embedding_row ) assert model assert dimensions > 0 # float32 = 4 bajty. assert ( blob_size == dimensions * 4 ) def test_database_rebuild_replaces_old_database_only_after_success( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: _, _, db_file = configure_indexer( monkeypatch, tmp_path, ) with sqlite3.connect( db_file ) as conn: conn.execute( """ CREATE TABLE marker( value TEXT ) """ ) conn.execute( """ INSERT INTO marker VALUES ('old database') """ ) conn.commit() def fail_validation( conn: sqlite3.Connection, ) -> None: raise RuntimeError( "úmyselná chyba validácie" ) monkeypatch.setattr( indexer, "validate_database", fail_validation, ) with pytest.raises( RuntimeError, match="úmyselná chyba", ): indexer.build_database() with sqlite3.connect( db_file ) as conn: assert ( conn.execute( """ SELECT value FROM marker """ ).fetchone()[0] == "old database" ) assert not ( indexer.temporary_database_path( db_file ).exists() ) def test_database_rejects_chunk_without_chunk_id( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: ( documents_file, chunks_file, _, ) = configure_indexer( monkeypatch, tmp_path, ) broken = sample_chunks() broken[0][ "chunk_id" ] = "" write_json( documents_file, sample_documents(), ) write_json( chunks_file, broken, ) with pytest.raises( ValueError, match="chunk_id", ): indexer.build_database()