180 lines
6.5 KiB
Python
180 lines
6.5 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Eval harness for insight fixtures (laptop/CI OK — no Ollama / Django required).
|
|
|
|
Runs CP707 grading + guardrail checks against fixtures/insight/*.json.
|
|
LLM narrate smoke belongs on the NUC (scripts/nuc_insight_smoke.sh).
|
|
|
|
Usage (from repo root):
|
|
python scripts/eval_insight_fixtures.py
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import re
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
ROOT = Path(__file__).resolve().parents[1]
|
|
sys.path.insert(0, str(ROOT / "backend" / "apps" / "operations" / "services"))
|
|
|
|
import cp707_knowledge as cp707 # noqa: E402
|
|
|
|
FIXTURES = ROOT / "fixtures" / "insight"
|
|
_NUMERIC_PIPE_ROW = re.compile(r"^[\d.,]+(\s*\|\s*[\d.,]*)+$")
|
|
|
|
|
|
def strip_headerless_tables(chunk: str) -> tuple[str, int]:
|
|
lines = str(chunk or "").split("\n")
|
|
removed = 0
|
|
kept: list[str] = []
|
|
for line in lines:
|
|
trimmed = line.strip()
|
|
if trimmed and _NUMERIC_PIPE_ROW.match(trimmed):
|
|
removed += 1
|
|
continue
|
|
kept.append(line)
|
|
if removed == 0:
|
|
return chunk, 0
|
|
text = "\n".join(kept).strip() + "\n[Tabel angka tanpa judul kolom dihapus]"
|
|
return text, removed
|
|
|
|
|
|
def load_fixtures() -> list[tuple[str, dict]]:
|
|
return [(p.name, json.loads(p.read_text(encoding="utf-8"))) for p in sorted(FIXTURES.glob("*.json"))]
|
|
|
|
|
|
def assert_true(cond: bool, msg: str, errors: list[str]) -> None:
|
|
if not cond:
|
|
errors.append(msg)
|
|
|
|
|
|
def eval_fixture(name: str, ctx: dict) -> list[str]:
|
|
errors: list[str] = []
|
|
assert_true(ctx.get("kandangId") is not None, f"{name}: kandangId wajib", errors)
|
|
|
|
day = ctx.get("hari_ke")
|
|
day_i = int(day) if isinstance(day, (int, float)) and day else None
|
|
|
|
if name.startswith("empty-context"):
|
|
analysis = cp707.analyze_with_cp707_standards(ctx)
|
|
assert_true(isinstance(analysis.get("analyses"), dict), f"{name}: analyses dict", errors)
|
|
# No numbers to invent — analyzers should simply omit or unknown
|
|
return errors
|
|
|
|
mort = ctx.get("mortalitas_kumulatif_persen")
|
|
if isinstance(mort, (int, float)) and mort >= 5:
|
|
block = cp707.analyze_mortality(float(mort), day_i or 28)
|
|
assert_true(
|
|
block["status"] in ("warning", "critical"),
|
|
f"{name}: mortalitas {mort}% harus warning/critical, got {block['status']}",
|
|
errors,
|
|
)
|
|
msg = block.get("message", "").lower()
|
|
assert_true(
|
|
"kritis" in msg or "waspada" in msg or "melebihi" in msg or ">" in msg,
|
|
f"{name}: pesan mortalitas harus tegas: {block.get('message')}",
|
|
errors,
|
|
)
|
|
|
|
if ctx.get("topic") == "fcr" and isinstance(ctx.get("fcr_terakhir"), (int, float)) and day_i:
|
|
block = cp707.analyze_fcr(float(ctx["fcr_terakhir"]), day_i)
|
|
assert_true("direction" in block, f"{name}: analyze_fcr harus punya direction", errors)
|
|
assert_true(
|
|
block.get("status") in ("ok", "warning", "critical", "unknown"),
|
|
f"{name}: status FCR invalid",
|
|
errors,
|
|
)
|
|
|
|
if ctx.get("report_type") == "end_cycle":
|
|
weeks = ctx.get("weekly_summaries")
|
|
assert_true(isinstance(weeks, list) and len(weeks) > 0, f"{name}: weekly_summaries wajib", errors)
|
|
|
|
sample = "prosa aman\n1 | 195 | 34 | 164,5 | 0,844\n2 | 499 | 50 | 530,5 | 1,063\n"
|
|
cleaned, removed = strip_headerless_tables(sample)
|
|
assert_true(removed >= 2, f"{name}: strip headerless tables", errors)
|
|
assert_true("164,5" not in cleaned.split("[")[0], f"{name}: angka tabel harus hilang dari prosa", errors)
|
|
|
|
# Gold mortality math: mati+afkir+panen == DOC
|
|
if name.startswith("gold-cycle"):
|
|
doc = ctx.get("doc_in_ekor")
|
|
mati = ctx.get("mati_ekor")
|
|
afkir = ctx.get("afkir_ekor")
|
|
panen = ctx.get("panen_ekor")
|
|
if all(isinstance(x, (int, float)) for x in (doc, mati, afkir, panen)):
|
|
assert_true(
|
|
int(mati) + int(afkir) + int(panen) == int(doc),
|
|
f"{name}: DOC harus = mati+afkir+panen (panen ≠ kematian)",
|
|
errors,
|
|
)
|
|
|
|
return errors
|
|
|
|
|
|
def main() -> int:
|
|
fixtures = load_fixtures()
|
|
if not fixtures:
|
|
print("FAIL: no fixtures found")
|
|
return 1
|
|
all_errors: list[str] = []
|
|
for name, ctx in fixtures:
|
|
errs = eval_fixture(name, ctx)
|
|
if errs:
|
|
print(f"FAIL {name}")
|
|
for e in errs:
|
|
print(f" - {e}")
|
|
all_errors.extend(errs)
|
|
else:
|
|
print(f"PASS {name}")
|
|
|
|
by_name = {n: c for n, c in fixtures}
|
|
assert_true(
|
|
by_name["kandang-01.json"]["kandangId"] != by_name["kandang-02.json"]["kandangId"],
|
|
"kandang-01/02 harus id berbeda",
|
|
all_errors,
|
|
)
|
|
assert_true(
|
|
by_name["kandang-01.json"]["fcr_terakhir"] != by_name["kandang-02.json"]["fcr_terakhir"],
|
|
"FCR antar kandang harus beda (anti bleed fixture)",
|
|
all_errors,
|
|
)
|
|
|
|
# Half-cycle baik vs buruk (mirrors DB seed Kandang 3 / 4)
|
|
if "half-good-h28.json" in by_name and "half-bad-h28.json" in by_name:
|
|
good = by_name["half-good-h28.json"]
|
|
bad = by_name["half-bad-h28.json"]
|
|
assert_true(good["kandangId"] != bad["kandangId"], "half good/bad id berbeda", all_errors)
|
|
assert_true(good["fcr_terakhir"] < bad["fcr_terakhir"], "half baik FCR < half buruk", all_errors)
|
|
assert_true(
|
|
good["mortalitas_persen"] < 5 <= bad["mortalitas_persen"],
|
|
"half baik mort <5%, buruk >=5%",
|
|
all_errors,
|
|
)
|
|
g_fcr = cp707.analyze_fcr(float(good["fcr_terakhir"]), 28)
|
|
b_fcr = cp707.analyze_fcr(float(bad["fcr_terakhir"]), 28)
|
|
assert_true(g_fcr["status"] == "ok", f"half-good FCR harus ok, got {g_fcr['status']}", all_errors)
|
|
assert_true(
|
|
b_fcr["status"] in ("warning", "critical"),
|
|
f"half-bad FCR harus warning/critical, got {b_fcr['status']}",
|
|
all_errors,
|
|
)
|
|
g_bw = cp707.analyze_bw(float(good["bobot_rata_rata_gram"]), 28)
|
|
b_bw = cp707.analyze_bw(float(bad["bobot_rata_rata_gram"]), 28)
|
|
assert_true(g_bw["status"] == "ok", f"half-good BW harus ok, got {g_bw['status']}", all_errors)
|
|
assert_true(
|
|
b_bw["status"] in ("warning", "critical"),
|
|
f"half-bad BW harus warning/critical, got {b_bw['status']}",
|
|
all_errors,
|
|
)
|
|
|
|
if all_errors:
|
|
print(f"\n{len(all_errors)} error(s)")
|
|
return 1
|
|
print(f"\nAll {len(fixtures)} fixtures OK (grading/guardrails). LLM generate: run on NUC.")
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|