87 lines
3.0 KiB
Python
87 lines
3.0 KiB
Python
#!/usr/bin/env python3
|
|
"""Приёмка recall@5 для memo (TESTING.md T2/T3).
|
|
|
|
Читает e2e/questions.tsv, гоняет CLI и считает, попал ли ожидаемый файл в top-5.
|
|
Использование:
|
|
python3 e2e/run_e2e.py --cli ./memo-cli/build/install/memo-cli/bin/memo \
|
|
--root /root/WORK/memo-e2e [--k 5] [--min-sem-recall 0.8]
|
|
"""
|
|
|
|
import argparse
|
|
import json
|
|
import shlex
|
|
import subprocess
|
|
import sys
|
|
|
|
|
|
def normalize(p: str, root: str) -> str:
|
|
p = p.replace("\\", "/")
|
|
root = root.rstrip("/")
|
|
if p.startswith(root + "/"):
|
|
p = p[len(root) + 1:]
|
|
return p.lstrip("./")
|
|
|
|
|
|
def main() -> int:
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--cli", required=True)
|
|
ap.add_argument("--root", required=True)
|
|
ap.add_argument("--questions", default="e2e/questions.tsv")
|
|
ap.add_argument("--k", type=int, default=5)
|
|
ap.add_argument("--min-sem-recall", type=float, default=0.8)
|
|
a = ap.parse_args()
|
|
|
|
rows = []
|
|
for line in open(a.questions, encoding="utf-8"):
|
|
line = line.rstrip("\n")
|
|
if not line or line.startswith("#"):
|
|
continue
|
|
q, scope, expected, kind = line.split("\t")
|
|
rows.append((q, scope, expected, kind))
|
|
|
|
hits = {"sem": 0, "lex": 0}
|
|
total = {"sem": 0, "lex": 0}
|
|
fails = []
|
|
for q, scope, expected, kind in rows:
|
|
total[kind] += 1
|
|
cmd = shlex.split(a.cli) + ["search", f"{a.root}/{scope}", q, "--k", str(a.k), "--json"]
|
|
try:
|
|
out = subprocess.run(cmd, capture_output=True, text=True, timeout=180)
|
|
except subprocess.TimeoutExpired:
|
|
fails.append((q, kind, "TIMEOUT", []))
|
|
continue
|
|
if out.returncode != 0:
|
|
fails.append((q, kind, f"exit {out.returncode}: {out.stderr.strip()[:200]}", []))
|
|
continue
|
|
try:
|
|
data = json.loads(out.stdout)
|
|
except json.JSONDecodeError:
|
|
fails.append((q, kind, f"не JSON: {out.stdout[:200]}", []))
|
|
continue
|
|
items = data if isinstance(data, list) else data.get("results", [])
|
|
got = [normalize(str(it.get("path", "")), a.root) for it in items][:a.k]
|
|
if normalize(expected, a.root) in got:
|
|
hits[kind] += 1
|
|
else:
|
|
fails.append((q, kind, "не найдено в top-%d" % a.k, got))
|
|
|
|
sem_total = total["sem"] or 1
|
|
lex_total = total["lex"] or 1
|
|
sem_recall = hits["sem"] / sem_total
|
|
lex_recall = hits["lex"] / lex_total
|
|
print(f"sem: {hits['sem']}/{sem_total} = {sem_recall:.2f} (порог {a.min_sem_recall})")
|
|
print(f"lex: {hits['lex']}/{lex_total} = {lex_recall:.2f} (порог 1.00)")
|
|
if fails:
|
|
print("\nпромахи:")
|
|
for q, kind, why, got in fails:
|
|
print(f" [{kind}] {q} -> {why}")
|
|
if got:
|
|
print(f" получено: {got}")
|
|
ok = sem_recall >= a.min_sem_recall and lex_recall >= 1.0
|
|
print("\nИТОГ:", "ПРОЙДЕНО" if ok else "ПРОВАЛ")
|
|
return 0 if ok else 1
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|