"""Tests for snapshot saving, baseline management and approval workflow.""" import json import pytest from pathlib import Path from scanner.baseline import BaselineManager, url_key def _sample_page(url: str, text: str = "Inhalt") -> dict: return { "url": url, "final_url": url, "status": 200, "content_type": "text/html", "response_headers": {}, "text": text, "links": {"a": [], "canonical": None, "meta_refresh": []}, "metadata": {"title": "T", "h1": [], "h2": []}, "hidden_content": [], "inline_scripts": [], "jsonld": [], "comment_links": [], } @pytest.fixture def bm(tmp_path: Path) -> BaselineManager: return BaselineManager(tmp_path / "data") class TestUrlKey: def test_deterministic(self): assert url_key("https://example.com/") == url_key("https://example.com/") def test_different_urls_different_keys(self): assert url_key("https://a.com/") != url_key("https://b.com/") def test_16_hex_chars(self): k = url_key("https://example.com/") assert len(k) == 16 assert all(c in "0123456789abcdef" for c in k) class TestSaveSnapshot: def test_creates_manifest(self, bm, tmp_path): pages = [_sample_page("https://b.info/")] snap_dir = bm.save_snapshot(pages) assert (snap_dir / "manifest.json").exists() def test_manifest_content(self, bm): pages = [_sample_page("https://b.info/"), _sample_page("https://b.info/p/")] snap_dir = bm.save_snapshot(pages) manifest = json.loads((snap_dir / "manifest.json").read_text()) assert manifest["page_count"] == 2 assert "https://b.info/" in manifest["urls"] def test_creates_page_files(self, bm): pages = [_sample_page("https://b.info/")] snap_dir = bm.save_snapshot(pages) page_files = list((snap_dir / "pages").glob("*.json")) assert len(page_files) == 1 def test_latest_snapshot_dir(self, bm): assert bm.latest_snapshot_dir() is None bm.save_snapshot([_sample_page("https://b.info/")]) assert bm.latest_snapshot_dir() is not None class TestLoadSnapshot: def test_loads_saved_snapshot(self, bm): pages = [_sample_page("https://b.info/"), _sample_page("https://b.info/p/")] bm.save_snapshot(pages) snap = bm.load_snapshot() assert "https://b.info/" in snap["pages"] assert "https://b.info/p/" in snap["pages"] def test_returns_empty_when_no_snapshot(self, bm): result = bm.load_snapshot() assert result == {} class TestBaselineExists: def test_false_before_approve(self, bm): assert not bm.baseline_exists() def test_true_after_approve(self, bm): bm.save_snapshot([_sample_page("https://b.info/")]) bm.approve_all(note="test") assert bm.baseline_exists() class TestApproveAll: def test_approve_all_creates_baseline(self, bm): pages = [_sample_page("https://b.info/"), _sample_page("https://b.info/p/")] bm.save_snapshot(pages) approved = bm.approve_all(note="Initial baseline") assert "https://b.info/" in approved assert "https://b.info/p/" in approved assert bm.baseline_exists() def test_manifest_contains_approval_metadata(self, bm): bm.save_snapshot([_sample_page("https://b.info/")]) bm.approve_all(note="test note", approved_by="testuser") manifest = json.loads((bm.baseline_dir / "manifest.json").read_text()) assert manifest["note"] == "test note" assert manifest["approved_by"] == "testuser" assert "approved_at" in manifest def test_approve_all_raises_without_snapshot(self, bm): with pytest.raises(RuntimeError): bm.approve_all() class TestApproveUrl: def test_approve_single_url(self, bm): pages = [_sample_page("https://b.info/"), _sample_page("https://b.info/p/")] bm.save_snapshot(pages) result = bm.approve_url("https://b.info/") assert result is True assert bm.baseline_exists() baseline = bm.load_baseline() assert "https://b.info/" in baseline["pages"] assert "https://b.info/p/" not in baseline["pages"] def test_approve_missing_url_returns_false(self, bm): bm.save_snapshot([_sample_page("https://b.info/")]) result = bm.approve_url("https://b.info/does-not-exist/") assert result is False class TestLoadBaseline: def test_loads_approved_pages(self, bm): pages = [_sample_page("https://b.info/", "Text A")] bm.save_snapshot(pages) bm.approve_all() baseline = bm.load_baseline() assert "https://b.info/" in baseline["pages"] assert baseline["pages"]["https://b.info/"]["text"] == "Text A" def test_never_auto_updates(self, bm): """Baseline must not change unless approve is explicitly called.""" bm.save_snapshot([_sample_page("https://b.info/", "Original")]) bm.approve_all() # Second snapshot with different content bm.save_snapshot([_sample_page("https://b.info/", "Manipuliert")]) # Baseline must still contain original content baseline = bm.load_baseline() assert baseline["pages"]["https://b.info/"]["text"] == "Original" class TestRebuildBaseline: def test_rebuild_replaces_baseline(self, bm): bm.save_snapshot([_sample_page("https://b.info/", "Alt")]) bm.approve_all() bm.save_snapshot([_sample_page("https://b.info/", "Neu")]) bm.rebuild_baseline(note="Major update") baseline = bm.load_baseline() assert baseline["pages"]["https://b.info/"]["text"] == "Neu" # --------------------------------------------------------------------------- # AI content inventory (SQLite store: verdict + description + url→hash) # --------------------------------------------------------------------------- def _entry(kind="image", url="https://b.info/x.jpg", category="clean", desc="Ein Logo"): return {"kind": kind, "url": url, "description": desc, "category": category, "severity": "none", "confidence": 0.9, "explanation": "ok", "model": "m", "dismissed": False, "checked_at": "2026-01-01T00:00:00"} class TestAiInventory: def test_roundtrip_preserves_entries_and_url_hashes(self, tmp_path): bm = BaselineManager(tmp_path) bm.save_ai_ledger({"entries": {"H1": _entry(desc="Schützenfest-Plakat")}, "url_hashes": {"https://b.info/x.jpg": "H1"}}) led = bm.load_ai_ledger() assert led["entries"]["H1"]["description"] == "Schützenfest-Plakat" assert led["entries"]["H1"]["dismissed"] is False assert led["url_hashes"] == {"https://b.info/x.jpg": "H1"} def test_migrates_json_ledger_once(self, tmp_path): (tmp_path).mkdir(parents=True, exist_ok=True) old = tmp_path / "ai_ledger.json" old.write_text(json.dumps({"entries": {"H1": _entry()}, "url_hashes": {"u": "H1"}}), encoding="utf-8") bm = BaselineManager(tmp_path) led = bm.load_ai_ledger() assert "H1" in led["entries"] assert not old.exists() # umbenannt assert (tmp_path / "ai_ledger.json.migrated").exists() assert (tmp_path / "content_inventory.db").exists() def test_first_seen_preserved_last_seen_updated(self, tmp_path): bm = BaselineManager(tmp_path) bm.save_ai_ledger({"entries": {"H1": _entry()}, "url_hashes": {}}) rows1 = bm.query_inventory() first1, last1 = rows1[0]["first_seen"], rows1[0]["last_seen"] import time; time.sleep(0.01) bm.save_ai_ledger({"entries": {"H1": _entry(category="propaganda")}, "url_hashes": {}}) rows2 = bm.query_inventory() assert rows2[0]["first_seen"] == first1 # bleibt assert rows2[0]["last_seen"] >= last1 # aktualisiert assert rows2[0]["category"] == "propaganda" # Feld aktualisiert def test_invalidate_removes_url(self, tmp_path): bm = BaselineManager(tmp_path) bm.save_ai_ledger({"entries": {}, "url_hashes": {"a": "H1", "b": "H2"}}) assert bm.invalidate_ai_url_hashes(["a"]) == 1 assert bm.load_ai_ledger()["url_hashes"] == {"b": "H2"} def test_query_search_and_csv_export(self, tmp_path): bm = BaselineManager(tmp_path) bm.save_ai_ledger({"entries": { "H1": _entry(url="https://b.info/kloster.jpg", desc="Foto des Klosters Bredelar"), "H2": _entry(url="https://b.info/auto.jpg", desc="Ein Auto"), }, "url_hashes": {}}) hits = bm.query_inventory(search="kloster") assert len(hits) == 1 and "Kloster" in hits[0]["description"] out = tmp_path / "inv.csv" assert bm.export_inventory_csv(out) == 2 content = out.read_text(encoding="utf-8") assert "description" in content and "Kloster" in content