252 lines
11 KiB
Python
252 lines
11 KiB
Python
"""
|
|
TOLARIA SEARCH SERVICE — C4B Test Suite
|
|
================================================================
|
|
Führt den eingefrorenen 22er-Ground-Truth-Corpus aus (alle exact/keyword/metadata-
|
|
Fälle), Acceptance-Tests, Quality-Metriken (Precision@k, Recall@k, MRR,
|
|
canonical-hit-rate, historical-error-rate, no-result-correctness), Latency
|
|
(P50/P95 für exact/keyword/metadata) und Failure-Tests.
|
|
|
|
Semantic-only Fälle (c09-c11, c14-c16, c22 mit mode=semantic/hybrid) sind in C4B
|
|
DEFERRED_TO_C4C / EXPECTED_NOT_IMPLEMENTED — sie zählen NICHT als C4B-Fail, wenn
|
|
der Honest-Mode-Contract korrekt erfüllt ist.
|
|
"""
|
|
import json
|
|
import sys
|
|
import os
|
|
import statistics
|
|
import time
|
|
|
|
sys.path.insert(0, os.path.dirname(__file__))
|
|
from search_api import TolariaSearch
|
|
from search_engine import INDEX_VERSION
|
|
|
|
_DIR = os.path.dirname(os.path.abspath(__file__))
|
|
SOURCE = os.path.join(_DIR, "index_source.json")
|
|
CORPUS = os.path.join(_DIR, "test_corpus_v1.1.json") # korrigierter, versionierter Corpus
|
|
|
|
|
|
def load_engine():
|
|
e = TolariaSearch()
|
|
e.rebuild_from_source(SOURCE, head=None, secret_filter=True)
|
|
return e
|
|
|
|
|
|
def validate_ground_truth(corpus, source):
|
|
"""VALIDATION STEP vor Testlauf. Alle object_id-Literale müssen real im
|
|
SoT-Index existieren und eindeutig sein. Bei Fehler => TEST SUITE BLOCKED."""
|
|
obj_ids = {o["id"] for o in source["objects"] if o.get("id")}
|
|
problems = []
|
|
for c in corpus["cases"]:
|
|
for key in ("expected_top", "expected_allowed", "must_not_top"):
|
|
for x in (c.get(key) or []):
|
|
if x.startswith("object/"):
|
|
if x not in obj_ids:
|
|
problems.append(f"{c['id']}.{key}: {x} nicht im SoT")
|
|
elif sum(1 for o in source["objects"] if o.get("id") == x) != 1:
|
|
problems.append(f"{c['id']}.{key}: {x} ambigue")
|
|
for k, v in corpus.get("verified_object_ids", {}).items():
|
|
if v.startswith("object/") and v not in obj_ids:
|
|
problems.append(f"verified.{k}: {v} nicht im SoT")
|
|
return problems
|
|
|
|
|
|
def run_case(engine, case):
|
|
mode = case.get("mode", "keyword")
|
|
flt = case.get("filters", {})
|
|
if not isinstance(flt, dict):
|
|
flt = {}
|
|
req = {"query": case.get("query", ""), "mode": mode,
|
|
"filters": flt,
|
|
"include_historical": flt.get("include_historical", False)}
|
|
return engine.search(req)
|
|
|
|
|
|
def main():
|
|
with open(CORPUS) as f:
|
|
corpus = json.load(f)
|
|
with open(SOURCE) as f:
|
|
source = json.load(f)
|
|
|
|
# --- VALIDATION STEP: Ground Truth muss gültig sein, sonst BLOCKED ---
|
|
vprobs = validate_ground_truth(corpus, source)
|
|
if vprobs:
|
|
print("GROUND_TRUTH_IDS_VALID = FALSE")
|
|
for p in vprobs:
|
|
print(" ", p)
|
|
print("TEST SUITE BLOCKED — kein Quality-PASS.")
|
|
sys.exit(2)
|
|
print("GROUND_TRUTH_IDS_VALID = TRUE")
|
|
|
|
engine = load_engine()
|
|
h = engine.health()
|
|
print(f"INDEX: version={h['index_version']} objects={h['object_count']} "
|
|
f"blocked={h['secret_blocked_objects']} head={h['source_head']}")
|
|
print(f"SUPPORTED_MODES={h['supported_modes']}")
|
|
print("=" * 80)
|
|
|
|
# --- 22er Corpus ---
|
|
results = []
|
|
for case in corpus["cases"]:
|
|
cid = case["id"]
|
|
mode = case.get("mode", "keyword")
|
|
if mode in ("semantic", "vector", "hybrid"):
|
|
# Honest-Mode prüfen
|
|
resp = engine.search({"query": case.get("query",""), "mode": mode,
|
|
"filters": case.get("filters", {})})
|
|
ok = (resp.get("requested_mode") == mode and
|
|
resp.get("actual_mode") is None and
|
|
resp.get("fallback") is False and
|
|
"mode_not_implemented" in str(resp.get("error", {}).get("code")))
|
|
results.append({"id": cid, "status": "PASS_NOT_IMPL" if ok else "FAIL",
|
|
"note": f"mode={mode} honest"})
|
|
continue
|
|
r = run_case(engine, case)
|
|
got_ids = [x["object_id"] for x in r.get("results", [])]
|
|
top = case.get("expected_top") or []
|
|
allowed = case.get("expected_allowed") or []
|
|
must_not = case.get("must_not_top") or []
|
|
status = "PASS"
|
|
reason = []
|
|
# no-result / secret-blocked case: 0 Treffer erwartet
|
|
if cid in ("c13-no-result", "c21-config-code"):
|
|
if len(r.get("results", [])) != 0:
|
|
status = "FAIL"
|
|
reason.append(f"erwartet 0 Treffer, aber {len(r.get('results', []))}")
|
|
else:
|
|
if top and not (set(top) & set(got_ids)):
|
|
status = "FAIL"
|
|
reason.append("expected_top fehlt")
|
|
if allowed and not (set(allowed) & set(got_ids)):
|
|
status = "FAIL"
|
|
reason.append("keine expected_allowed gefunden")
|
|
if must_not and got_ids and (set(must_not) & set(got_ids[:1])):
|
|
status = "FAIL"
|
|
reason.append("must_not_top in Top-1")
|
|
results.append({"id": cid, "status": status, "got_ids": got_ids[:5],
|
|
"reason": "; ".join(reason) if reason else "", "total": r.get("total")})
|
|
|
|
print("=== GROUND-TRUTH (22er Corpus) ===")
|
|
for r in results:
|
|
print(f" [{r['status']:14}] {r['id']:28} total={r.get('total')} "
|
|
f"{('reason: '+r['reason']) if r.get('reason') else ''}")
|
|
print("=" * 80)
|
|
|
|
# --- Acceptance (Auswahl) ---
|
|
print("=== ACCEPTANCE (repräsentativ) ===")
|
|
acc = [
|
|
("exact-title", {"query":"Modul-09-Execution-Service","mode":"exact"}),
|
|
("keyword", {"query":"Intrabar Execution Gap","mode":"keyword"}),
|
|
("phrase", {"query":"deterministischer Execution-Service","mode":"keyword"}),
|
|
("metadata-only", {"query":"","mode":"metadata","filters":{"type":"arch","role":"module"}}),
|
|
("canonical/source", {"query":"Infrastructure","mode":"keyword"}),
|
|
("current/historical", {"query":"M12 DatasetGate","mode":"keyword","include_historical":False}),
|
|
("README-explicit", {"query":"README","mode":"keyword"}),
|
|
("overview-downrank", {"query":"Infrastruktur Betrieb","mode":"keyword"}),
|
|
("module-number", {"query":"Modul 15","mode":"keyword"}),
|
|
("ticker-acronym", {"query":"OHLCV","mode":"keyword"}),
|
|
("env-config", {"query":"POSTGRES_DB","mode":"keyword","filters":{"type":"code"}}),
|
|
("no-result", {"query":"xyzzy-foobar-42","mode":"keyword"}),
|
|
("historical-explicit", {"query":"historical phase8","mode":"keyword","include_historical":True}),
|
|
("source-only", {"query":"source infrastructure","mode":"keyword","filters":{"representation":"source"}}),
|
|
("canonical-only", {"query":"infrastructure","mode":"keyword","filters":{"representation":"canonical"}}),
|
|
("secret-blocked", {"query":"OLLAMA_API_KEY","mode":"keyword","filters":{"type":"code"}}),
|
|
]
|
|
for name, req in acc:
|
|
r = engine.search(req)
|
|
got = [x["object_id"] or x["path"] for x in r.get("results", [])]
|
|
print(f" {name:20} top={got[:4] if got else '∅'} total={r.get('total')}")
|
|
|
|
print("=" * 80)
|
|
|
|
# --- Quality Metrics (Keyword-fähige Fälle) ---
|
|
print("=== QUALITY METRICS ===")
|
|
metric_cases = [c for c in corpus["cases"] if c.get("mode") in ("exact","keyword")]
|
|
n = len(metric_cases)
|
|
mrr = 0.0
|
|
hits5 = 0
|
|
hits10 = 0
|
|
canonical_top = 0
|
|
hist_err = 0
|
|
no_result_ok = 0
|
|
total_canonical = 0
|
|
total_hist_checked = 0
|
|
for c in metric_cases:
|
|
flt = c.get("filters", {})
|
|
if not isinstance(flt, dict):
|
|
flt = {}
|
|
r = engine.search({"query": c.get("query", ""), "mode": c.get("mode", "keyword"),
|
|
"filters": flt,
|
|
"include_historical": flt.get("include_historical", False)})
|
|
got = [x["object_id"] for x in r.get("results",[])]
|
|
allowed = set(c.get("expected_allowed") or [])
|
|
top = c.get("expected_top") or []
|
|
# MRR
|
|
for rank, g in enumerate(got, 1):
|
|
if g in allowed or (top and g in top):
|
|
mrr += 1/rank
|
|
break
|
|
# Recall@5 / @10
|
|
relevant = allowed | set(top)
|
|
if not relevant:
|
|
if c.get("expected_top") == [] and len(got) == 0:
|
|
no_result_ok += 1
|
|
continue
|
|
hits5 += (len(set(got[:5]) & relevant) > 0)
|
|
hits10 += (len(set(got[:10]) & relevant) > 0)
|
|
# canonical-hit-rate
|
|
if c.get("expected_rep") == "canonical":
|
|
total_canonical += 1
|
|
if got and (r["results"][0].get("representation") == "canonical"):
|
|
canonical_top += 1
|
|
# historical-error-rate
|
|
if not flt.get("include_historical"):
|
|
total_hist_checked += 1
|
|
if got and r["results"][0].get("state") in ("historical","superseded","archived"):
|
|
hist_err += 1
|
|
print(f" MRR = {round(mrr/n,4)} (n={n})")
|
|
print(f" Recall@5 = {round(hits5/n,3)} Recall@10 = {round(hits10/n,3)}")
|
|
print(f" canonical-hit-rate = {round(canonical_top/max(1,total_canonical),3)} ({canonical_top}/{total_canonical})")
|
|
print(f" historical-error-rate = {round(hist_err/max(1,total_hist_checked),3)} ({hist_err}/{total_hist_checked})")
|
|
print(f" no-result-correctness = {no_result_ok}")
|
|
|
|
print("=" * 80)
|
|
|
|
# --- Latency (echte Messungen) ---
|
|
print("=== LATENCY (ms, gemessen) ===")
|
|
lat = {"exact": [], "keyword": [], "metadata": []}
|
|
for _ in range(200):
|
|
for mode, q in [("exact","Modul-09"), ("keyword","Infrastructure OHLCV"), ("metadata","")]:
|
|
t0 = time.time()
|
|
engine.search({"query": q, "mode": mode, "filters": {} if mode!="metadata" else {"type":"arch"}})
|
|
lat[mode].append((time.time()-t0)*1000)
|
|
for mode, vals in lat.items():
|
|
vals.sort()
|
|
p50 = vals[len(vals)//2]
|
|
p95 = vals[int(len(vals)*0.95)]
|
|
print(f" {mode:10} P50={round(p50,2)}ms P95={round(p95,2)}ms")
|
|
|
|
print("=" * 80)
|
|
|
|
# --- Failure-Tests ---
|
|
print("=== FAILURE TESTS ===")
|
|
fail = [
|
|
("empty query (keyword)", {"query":"","mode":"keyword"}),
|
|
("invalid mode", {"query":"x","mode":"nonsense"}),
|
|
("unsupported semantic", {"query":"x","mode":"semantic"}),
|
|
("unsupported vector", {"query":"x","mode":"vector"}),
|
|
("unsupported hybrid", {"query":"x","mode":"hybrid"}),
|
|
("limit>100", {"query":"x","mode":"keyword","limit":200}),
|
|
("no-result", {"query":"xyzzy-foobar-42","mode":"keyword"}),
|
|
]
|
|
for name, req in fail:
|
|
r = engine.search(req)
|
|
if "error" in r:
|
|
print(f" {name:28} → error={r['error']['code']}")
|
|
elif r.get("actual_mode") is None:
|
|
print(f" {name:28} → honest: requested={r['requested_mode']} actual={r['actual_mode']}")
|
|
else:
|
|
print(f" {name:28} → results={r['total']}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|