ai insight rework #1
This commit is contained in:
1 parent
0d960c8153
commit
9d76f54d33
13 files changed
+730
-299
No files matched your search
@@ -10,7 +10,11 @@ from __future__ import annotations
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
from typing import Any
|
||||
import threading
|
||||
import time
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable
|
||||
|
||||
import httpx
|
||||
from django.conf import settings
|
||||
@@ -23,6 +27,9 @@ from apps.operations.services import root_cause
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Plan §5: one LLM inference at a time (CPU server; prevents thread thrash).
|
||||
_INFERENCE_LOCK = threading.Lock()
|
||||
|
||||
_NUMERIC_PIPE_ROW = re.compile(r"^[\d.,]+(\s*\|\s*[\d.,]*)+$")
|
||||
|
||||
# Stored in AIInsight.alert — condition of graded data, not narrative text.
|
||||
@@ -434,7 +441,7 @@ def fetch_rag_chunks(query: str, topic: str, n_results: int = 4) -> tuple[list[s
|
||||
return [], []
|
||||
|
||||
|
||||
def call_ollama(system_prompt: str, user_prompt: str) -> str | None:
|
||||
def call_ollama(system_prompt: str, user_prompt: str) -> tuple[str | None, dict[str, Any]]:
|
||||
model = getattr(settings, "LLM_MODEL_NAME", "qwen2.5:3b") or "qwen2.5:3b"
|
||||
base = getattr(settings, "OLLAMA_BASE_URL", "http://127.0.0.1:11434") or "http://127.0.0.1:11434"
|
||||
url = f"{base.rstrip('/')}/api/chat"
|
||||
@@ -442,7 +449,15 @@ def call_ollama(system_prompt: str, user_prompt: str) -> str | None:
|
||||
payload = {
|
||||
"model": model,
|
||||
"stream": False,
|
||||
"options": {"temperature": 0.2},
|
||||
# format:"json" (grammar) keeps the model from rambling in prose; the schema
|
||||
# reminder appended AFTER the page-data JSON keeps it from copying that JSON.
|
||||
"format": "json",
|
||||
"options": {
|
||||
"temperature": float(getattr(settings, "LLM_TEMPERATURE", 0.2) or 0.2),
|
||||
"seed": int(getattr(settings, "LLM_SEED", 42) or 42),
|
||||
"num_ctx": int(getattr(settings, "LLM_NUM_CTX", 2048) or 2048),
|
||||
"num_thread": int(getattr(settings, "LLM_NUM_THREAD", 4) or 4),
|
||||
},
|
||||
"messages": [
|
||||
{"role": "system", "content": system_prompt},
|
||||
{"role": "user", "content": user_prompt},
|
||||
@@ -450,16 +465,23 @@ def call_ollama(system_prompt: str, user_prompt: str) -> str | None:
|
||||
}
|
||||
try:
|
||||
with httpx.Client(timeout=timeout) as client:
|
||||
resp = client.post(url, json=payload)
|
||||
# Plan §5: serialize LLM calls so concurrent generates don't thrash CPU.
|
||||
with _INFERENCE_LOCK:
|
||||
resp = client.post(url, json=payload)
|
||||
if resp.status_code >= 400:
|
||||
logger.error("Ollama HTTP %s: %s", resp.status_code, resp.text[:500])
|
||||
return None
|
||||
return None, {}
|
||||
data = resp.json()
|
||||
message = data.get("message") or {}
|
||||
return message.get("content") or data.get("response")
|
||||
content = message.get("content") or data.get("response")
|
||||
usage = {
|
||||
"prompt_tokens": data.get("prompt_eval_count"),
|
||||
"completion_tokens": data.get("eval_count"),
|
||||
}
|
||||
return content, usage
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.error("Ollama call failed: %s", exc)
|
||||
return None
|
||||
return None, {}
|
||||
|
||||
|
||||
def parse_llm_json(text: str) -> dict[str, str] | None:
|
||||
@@ -530,23 +552,6 @@ def parse_end_cycle_llm_json(text: str) -> dict[str, Any] | None:
|
||||
return parsed
|
||||
|
||||
|
||||
def local_fallback_insight(graded: dict[str, Any], topic: str) -> dict[str, str]:
|
||||
analyses = graded.get("analyses") or {}
|
||||
lines = []
|
||||
for name, block in analyses.items():
|
||||
if isinstance(block, dict) and block.get("message"):
|
||||
lines.append(f"{name}: {block['message']}")
|
||||
if not lines:
|
||||
return {
|
||||
"kesimpulan": f"Data untuk topik {topic} tidak cukup untuk dianalisis (status unknown).",
|
||||
"insight": "Lengkapi data operasional kandang, lalu generate ulang. Jangan mengarang angka.",
|
||||
}
|
||||
return {
|
||||
"kesimpulan": lines[0],
|
||||
"insight": "\n".join(lines[1:]) if len(lines) > 1 else lines[0],
|
||||
}
|
||||
|
||||
|
||||
_SOFT_MORTALITY_WORD = re.compile(
|
||||
r"\b(rendah|normal|baik|aman|terkendali|sedikit)\b",
|
||||
re.IGNORECASE,
|
||||
@@ -616,6 +621,11 @@ def collapse_duplicate_kandang(text: str) -> str:
|
||||
return _DUP_KANDANG.sub(lambda m: "Kandang" if m.group(0)[0].isupper() else "kandang", text)
|
||||
|
||||
|
||||
def _placeholder_text(value: str) -> bool:
|
||||
"""True when a narrative field is empty / schema placeholder (e.g. '...')."""
|
||||
return len(value.strip().strip(".…<>- \t")) < 20
|
||||
|
||||
|
||||
def sanitize_insight_narrative(parsed: dict[str, Any]) -> dict[str, Any]:
|
||||
"""Deterministic cleanup of narrative fields after the LLM."""
|
||||
out = dict(parsed)
|
||||
@@ -736,6 +746,50 @@ def serialize_insight(row: AIInsight) -> dict[str, Any]:
|
||||
}
|
||||
|
||||
|
||||
# Plan Stage 1: graded status -> deterministic RAG keywords (no LLM query rewrite).
|
||||
_DIAGNOSTIC_KEYWORDS = {
|
||||
"mortality": "penanganan mortalitas kumulatif deplesi penyebab kematian brooding",
|
||||
"fcr": "target FCR pakan harian konsumsi pakan efisiensi ventilasi",
|
||||
"bw": "target bobot badan ADG pertumbuhan standar mingguan",
|
||||
"iot": "standar ventilasi suhu kelembapan amonia sekam basah kandang",
|
||||
"environment": "standar ventilasi suhu kelembapan amonia sekam basah kandang",
|
||||
}
|
||||
|
||||
|
||||
def diagnose_rag_keywords(graded: dict[str, Any]) -> str:
|
||||
"""Stage 1: anomaly keywords from graded blocks; '' when everything ok/unknown."""
|
||||
analyses = (graded or {}).get("analyses") or {}
|
||||
clues: list[str] = []
|
||||
for key, block in analyses.items():
|
||||
if not isinstance(block, dict):
|
||||
continue
|
||||
if str(block.get("status") or "").lower() in ("warning", "critical"):
|
||||
hint = _DIAGNOSTIC_KEYWORDS.get(key)
|
||||
if hint and hint not in clues:
|
||||
clues.append(hint)
|
||||
return " ".join(clues)
|
||||
|
||||
|
||||
def _notify(stage_cb: Callable[[str], None] | None, stage: str) -> None:
|
||||
if not stage_cb:
|
||||
return
|
||||
try:
|
||||
stage_cb(stage)
|
||||
except Exception: # noqa: BLE001 - progress UI must never break generation
|
||||
logger.debug("stage_cb(%s) failed", stage, exc_info=True)
|
||||
|
||||
|
||||
def _log_trace(record: dict[str, Any]) -> None:
|
||||
"""Plan Stage 5: append-only audit trail for replay/debug."""
|
||||
try:
|
||||
path = Path(settings.BASE_DIR) / "logs" / "rag_trace.jsonl"
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with path.open("a", encoding="utf-8") as fh:
|
||||
fh.write(json.dumps(record, ensure_ascii=False, default=str) + "\n")
|
||||
except Exception: # noqa: BLE001 - logging must never fail the request
|
||||
logger.warning("rag_trace write failed", exc_info=True)
|
||||
|
||||
|
||||
def generate_insight(
|
||||
*,
|
||||
cycle_id: int,
|
||||
@@ -745,6 +799,7 @@ def generate_insight(
|
||||
report_type: str = "page",
|
||||
report_period: str = "current",
|
||||
force_refresh: bool = False,
|
||||
stage_cb: Callable[[str], None] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
if not kandang_id:
|
||||
raise ValueError("kandang_id wajib — insight tidak boleh untuk semua kandang")
|
||||
@@ -785,7 +840,11 @@ def generate_insight(
|
||||
ctx.setdefault("cycleId", cycle_id)
|
||||
ctx = prune_context_by_period(ctx, report_type, report_period)
|
||||
|
||||
t_start = time.monotonic()
|
||||
_notify(stage_cb, "grading")
|
||||
graded = grade_context(ctx)
|
||||
t_after_grade = time.monotonic()
|
||||
_notify(stage_cb, "retrieving")
|
||||
day = graded.get("hari_ke")
|
||||
standard_block = cp707.build_cp707_standard_block(ctx) or cp707.build_book_reference_context(day or 1)
|
||||
|
||||
@@ -805,7 +864,10 @@ def generate_insight(
|
||||
)
|
||||
else:
|
||||
rag_query = f"panduan manajemen broiler CP 707 untuk {topic} umur hari ke-{day or '?'}"
|
||||
rag_query = f"{rag_query} {diagnose_rag_keywords(graded)}".strip()
|
||||
chunks, citations = fetch_rag_chunks(rag_query, topic)
|
||||
t_after_rag = time.monotonic()
|
||||
_notify(stage_cb, "synthesizing")
|
||||
|
||||
system_prompt = (
|
||||
"Anda adalah asisten farm broiler on-premise. Tugas Anda HANYA menulis narasi "
|
||||
@@ -829,8 +891,12 @@ def generate_insight(
|
||||
f"\n[NEXT CYCLE ACTIONS]\n"
|
||||
f"{json.dumps(next_actions, ensure_ascii=False, default=str)}\n"
|
||||
)
|
||||
cited_chunks = [
|
||||
f"[cp707#chunk{(citations[i].get('chunk_id') if i < len(citations) else None) or i}] {chunk}"
|
||||
for i, chunk in enumerate(chunks[:4])
|
||||
]
|
||||
if chunks:
|
||||
system_prompt += "\n[CUPLIKAN SOP CP 707 — prosa]\n" + "\n---\n".join(chunks[:4])
|
||||
system_prompt += "\n[CUPLIKAN SOP CP 707 — prosa]\n" + "\n---\n".join(cited_chunks)
|
||||
|
||||
user_prompt = build_insight_user_prompt(
|
||||
topic=topic,
|
||||
@@ -839,44 +905,152 @@ def generate_insight(
|
||||
report_period=report_period,
|
||||
context=ctx,
|
||||
)
|
||||
# Page-data JSON ends the user message; small models copy the last JSON they
|
||||
# read. Re-state the required schema AFTER the data so recency wins.
|
||||
# Status/facts go right before it: book chunks (system tail) describe IDEAL
|
||||
# conditions, so without this last the model narrates ideals over actual status.
|
||||
# Skip "unknown" blocks when at least one block has a real status:
|
||||
# no-data lines are padding triggers (e.g. mortality spun into FCR
|
||||
# narrative). All-unknown case keeps them: nothing factual to lean on
|
||||
# and "data tidak tersedia" is then the only correct narration.
|
||||
known_lines = [
|
||||
f"{name} = {block.get('status')}: {block.get('message')}"
|
||||
for name, block in (graded.get("analyses") or {}).items()
|
||||
if isinstance(block, dict)
|
||||
and block.get("message")
|
||||
and block.get("status") != "unknown"
|
||||
]
|
||||
unknown_lines = [
|
||||
f"{name} = {block.get('status')}: {block.get('message')}"
|
||||
for name, block in (graded.get("analyses") or {}).items()
|
||||
if isinstance(block, dict)
|
||||
and block.get("message")
|
||||
and block.get("status") == "unknown"
|
||||
]
|
||||
status_lines = known_lines or unknown_lines
|
||||
actual_alert = alert_from_graded(graded)
|
||||
user_prompt = (user_prompt or "").rstrip() + (
|
||||
"\nKONTEKS: semua data adalah ternak AYAM BROILER (unggas) di kandang — "
|
||||
"BUKAN tanaman/pertanian. 'Panen' = panen ayam. 'Bobot' = gram per ekor "
|
||||
"(bukan per karung). 'ADG' = kenaikan bobot harian ayam."
|
||||
f"\nSTATUS AKTUAL: {actual_alert}. Kesimpulan dan insight HARUS konsisten "
|
||||
"dengan status ini: bila warning/critical, sebut masalahnya dan bandingkan "
|
||||
"dengan angka standar CP 707; dilarang menyebut ideal/sehat/baik/aman "
|
||||
"bila STATUS AKTUAL bukan healthy."
|
||||
+ ("\nFAKTA GRADED:\n- " + "\n- ".join(status_lines) if status_lines else "")
|
||||
# No length req anywhere -> 3B model sometimes stops after 1 sentence
|
||||
# (three EEF rows stored identical 136/127 chars). Demand structure.
|
||||
+ "\nPANJANG WAJIB: insight = narasi 3-6 kalimat berurutan: "
|
||||
"(1) angka aktual vs standar CP 707, (2) penyebab/implikasi, "
|
||||
"(3) tindakan konkret. kesimpulan = 1-3 kalimat. "
|
||||
"Satu kalimat singkat = gagal."
|
||||
# Placeholder <> (not "..."): model has copied schema "..." verbatim
|
||||
# into insight, which stored fine and rendered as an empty page.
|
||||
+ '\nBALAS HANYA JSON persis: {"kesimpulan":"<ringkasan>","insight":"<narasi panjang>"}'
|
||||
+ ' - JANGAN salin teks <> dari contoh; tulis kalimat lengkap sendiri.'
|
||||
)
|
||||
|
||||
raw = call_ollama(system_prompt, user_prompt)
|
||||
if is_end_cycle:
|
||||
parsed_raw = parse_end_cycle_llm_json(raw or "")
|
||||
if parsed_raw:
|
||||
# Apply mortality wording on narrative fields before sanitize.
|
||||
soft = {
|
||||
"kesimpulan": str(parsed_raw.get("kesimpulan") or ""),
|
||||
"insight": str(parsed_raw.get("insight") or ""),
|
||||
}
|
||||
soft = enforce_mortality_wording(soft, graded)
|
||||
soft = sanitize_insight_narrative(soft)
|
||||
parsed_raw["kesimpulan"] = soft.get("kesimpulan")
|
||||
parsed_raw["insight"] = soft.get("insight")
|
||||
payload = root_cause.sanitize_end_cycle_payload(
|
||||
parsed_raw,
|
||||
hypotheses=hypotheses,
|
||||
actions=next_actions,
|
||||
masalah=masalah,
|
||||
# Re-state output schema as the system prompt's final line (user_prompt
|
||||
# repeats it again after the page-data JSON).
|
||||
system_prompt += (
|
||||
"\nINGAT: keluaran HANYA JSON valid sesuai FORMAT OUTPUT di atas "
|
||||
"(wajib ada key kesimpulan dan insight). Tanpa teks lain."
|
||||
)
|
||||
# Degenerate output (schema placeholder like "..." copied verbatim) stored
|
||||
# fine as JSON and rendered as an empty insight page. Reject + retry once,
|
||||
# then fail loud (same policy as parse failure).
|
||||
degenerate_reason: str | None = None
|
||||
for attempt in range(2):
|
||||
raw, usage = call_ollama(system_prompt, user_prompt)
|
||||
t_after_llm = time.monotonic()
|
||||
if is_end_cycle:
|
||||
parsed_raw = parse_end_cycle_llm_json(raw or "")
|
||||
if parsed_raw:
|
||||
# Apply mortality wording on narrative fields before sanitize.
|
||||
soft = {
|
||||
"kesimpulan": str(parsed_raw.get("kesimpulan") or ""),
|
||||
"insight": str(parsed_raw.get("insight") or ""),
|
||||
}
|
||||
soft = enforce_mortality_wording(soft, graded)
|
||||
soft = sanitize_insight_narrative(soft)
|
||||
parsed_raw["kesimpulan"] = soft.get("kesimpulan")
|
||||
parsed_raw["insight"] = soft.get("insight")
|
||||
payload = root_cause.sanitize_end_cycle_payload(
|
||||
parsed_raw,
|
||||
hypotheses=hypotheses,
|
||||
actions=next_actions,
|
||||
masalah=masalah,
|
||||
)
|
||||
payload = sanitize_insight_narrative(payload)
|
||||
else:
|
||||
# Fallback removed (2026-09-25): fail loudly instead of serving fake insight.
|
||||
if not raw:
|
||||
raise RuntimeError("Ollama tidak mengembalikan output apa pun (cek layanan LLM).")
|
||||
raise RuntimeError(
|
||||
"Output LLM end-cycle tidak sesuai format JSON yang diminta: "
|
||||
+ repr(str(raw)[:200])
|
||||
)
|
||||
summary = collapse_duplicate_kandang(str(payload.get("kesimpulan") or ""))
|
||||
insight_text = collapse_duplicate_kandang(
|
||||
root_cause.format_end_cycle_insight_text(payload)
|
||||
)
|
||||
payload = sanitize_insight_narrative(payload)
|
||||
else:
|
||||
payload = root_cause.local_fallback_end_cycle(graded, ctx)
|
||||
summary = collapse_duplicate_kandang(str(payload.get("kesimpulan") or ""))
|
||||
insight_text = collapse_duplicate_kandang(
|
||||
root_cause.format_end_cycle_insight_text(payload)
|
||||
)
|
||||
else:
|
||||
parsed = parse_llm_json(raw or "")
|
||||
if not parsed:
|
||||
parsed = local_fallback_insight(graded, topic)
|
||||
else:
|
||||
parsed = parse_llm_json(raw or "")
|
||||
if not parsed:
|
||||
# Fallback removed (2026-09-25): fail loudly instead of serving fake insight.
|
||||
if not raw:
|
||||
raise RuntimeError("Ollama tidak mengembalikan output apa pun (cek layanan LLM).")
|
||||
raise RuntimeError(
|
||||
"Output LLM tidak sesuai format JSON yang diminta: " + repr(str(raw)[:200])
|
||||
)
|
||||
parsed = enforce_mortality_wording(parsed, graded)
|
||||
parsed = sanitize_insight_narrative(parsed)
|
||||
insight_text = parsed["insight"]
|
||||
summary = parsed["kesimpulan"]
|
||||
parsed = sanitize_insight_narrative(parsed)
|
||||
insight_text = parsed["insight"]
|
||||
summary = parsed["kesimpulan"]
|
||||
if _placeholder_text(summary) or _placeholder_text(insight_text):
|
||||
degenerate_reason = f"kesimpulan={summary!r} insight={insight_text!r}"
|
||||
user_prompt += (
|
||||
"\nCATATAN: keluaran sebelumnya hanya placeholder. "
|
||||
"Tulis kalimat lengkap sendiri untuk kesimpulan dan insight."
|
||||
)
|
||||
continue
|
||||
degenerate_reason = None
|
||||
break
|
||||
if degenerate_reason:
|
||||
raise RuntimeError(
|
||||
"Output LLM degenerate (placeholder bukan kalimat): "
|
||||
+ degenerate_reason[:200]
|
||||
)
|
||||
|
||||
alert = alert_from_graded(graded)
|
||||
_notify(stage_cb, "saving")
|
||||
_log_trace(
|
||||
{
|
||||
"timestamp": timezone.now().isoformat(),
|
||||
"request_id": str(uuid.uuid4()),
|
||||
"cycle_id": cycle_id,
|
||||
"kandang_name": kandang.kandang_name,
|
||||
"topic": topic,
|
||||
"report_type": report_type,
|
||||
"report_period": report_period,
|
||||
"model": getattr(settings, "LLM_MODEL_NAME", None),
|
||||
"prompt_version": "v2.1",
|
||||
"formulated_query": rag_query,
|
||||
"retrieved_chunk_ids": [
|
||||
f"cp707#{c.get('chunk_id')}" if c.get("chunk_id") is not None else f"cp707#{c.get('source')}"
|
||||
for c in citations
|
||||
],
|
||||
"graded_alert": alert,
|
||||
"llm_ok": raw is not None,
|
||||
"latency_breakdown": {
|
||||
"grade_ms": round((t_after_grade - t_start) * 1000),
|
||||
"retrieval_ms": round((t_after_rag - t_after_grade) * 1000),
|
||||
"llm_ms": round((t_after_llm - t_after_rag) * 1000),
|
||||
"total_ms": round((time.monotonic() - t_start) * 1000),
|
||||
},
|
||||
"token_usage": usage,
|
||||
}
|
||||
)
|
||||
|
||||
row, _created = AIInsight.objects.update_or_create(
|
||||
cycle=cycle,
|
||||
|
||||
Reference in new issue
Block a user