ai insight rework #1
This commit is contained in:
1 parent
0d960c8153
commit
9d76f54d33
13 files changed
+730
-299
No files matched your search
@@ -0,0 +1,30 @@
|
||||
"""Drop orphan schema left by deleted migration 0019_aiinsight_structured_data_and_task.
|
||||
|
||||
0019 was applied to production Postgres (added ai_insight.structured_data NOT NULL +
|
||||
ai_insight_task table) but its file and the corresponding model fields were after-
|
||||
wards removed from the codebase. Django builds schema state from migration files,
|
||||
so INSERTs omit `structured_data` -> every AIInsight save fails with
|
||||
"null value in column structured_data violates not-null constraint".
|
||||
|
||||
Postgres-only: SQLite dev databases never had 0019 applied. Reverse is a no-op.
|
||||
"""
|
||||
from django.db import migrations
|
||||
|
||||
|
||||
def drop_orphan_schema(apps, schema_editor):
|
||||
if schema_editor.connection.vendor != "postgresql":
|
||||
return
|
||||
with schema_editor.connection.cursor() as cursor:
|
||||
cursor.execute("ALTER TABLE ai_insight DROP COLUMN IF EXISTS structured_data")
|
||||
cursor.execute("DROP TABLE IF EXISTS ai_insight_task")
|
||||
|
||||
|
||||
class Migration(migrations.Migration):
|
||||
|
||||
dependencies = [
|
||||
("operations", "0018_remove_aiinsight_source"),
|
||||
]
|
||||
|
||||
operations = [
|
||||
migrations.RunPython(drop_orphan_schema, migrations.RunPython.noop),
|
||||
]
|
||||
@@ -0,0 +1,107 @@
|
||||
"""Phase 4 async job runner for POST /ai-insights/generate/ (202 + polling).
|
||||
|
||||
Jobs persist as one JSON file each under `logs/insight_jobs/` so any gunicorn
|
||||
worker can answer the poll (ponytail: single-host only — move to Redis/DB queue
|
||||
if API ever spans multiple machines). Only the owning worker writes a job file,
|
||||
so no cross-process locking is needed; writes are atomic (tmp + os.replace).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import threading
|
||||
import time
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from django.conf import settings
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_JOBS_DIR = Path(settings.BASE_DIR) / "logs" / "insight_jobs"
|
||||
_MAX_AGE_SECONDS = 24 * 3600
|
||||
|
||||
|
||||
def _job_path(job_id: str) -> Path:
|
||||
return _JOBS_DIR / f"{job_id}.json"
|
||||
|
||||
|
||||
def _write(job_id: str, job: dict[str, Any]) -> None:
|
||||
tmp = _job_path(job_id).with_suffix(".tmp")
|
||||
tmp.write_text(json.dumps(job, ensure_ascii=False, default=str), encoding="utf-8")
|
||||
os.replace(tmp, _job_path(job_id))
|
||||
|
||||
|
||||
def _prune() -> None:
|
||||
"""Drop job files older than 24h so the dir stays bounded."""
|
||||
try:
|
||||
cutoff = time.time() - _MAX_AGE_SECONDS
|
||||
for f in _JOBS_DIR.glob("*.json"):
|
||||
if f.stat().st_mtime < cutoff:
|
||||
f.unlink(missing_ok=True)
|
||||
except OSError:
|
||||
logger.debug("insight_jobs prune failed", exc_info=True)
|
||||
|
||||
|
||||
def create_job(params: dict[str, Any]) -> str:
|
||||
"""Persist job record and start worker thread. Returns job_id for polling."""
|
||||
job_id = str(uuid.uuid4())
|
||||
_JOBS_DIR.mkdir(parents=True, exist_ok=True)
|
||||
_prune()
|
||||
_write(
|
||||
job_id,
|
||||
{
|
||||
"id": job_id,
|
||||
"status": "queued",
|
||||
"stage": "queued",
|
||||
"result": None,
|
||||
"error": None,
|
||||
},
|
||||
)
|
||||
threading.Thread(target=_run, args=(job_id, params), daemon=True).start()
|
||||
return job_id
|
||||
|
||||
|
||||
def get_job(job_id: str) -> dict[str, Any] | None:
|
||||
"""Read job from shared dir — works from any worker process."""
|
||||
try:
|
||||
raw = _job_path(job_id).read_text(encoding="utf-8")
|
||||
except FileNotFoundError:
|
||||
return None
|
||||
except OSError:
|
||||
logger.warning("insight_jobs read failed for %s", job_id, exc_info=True)
|
||||
return None
|
||||
try:
|
||||
job = json.loads(raw)
|
||||
except json.JSONDecodeError:
|
||||
return None
|
||||
return job if isinstance(job, dict) else None
|
||||
|
||||
|
||||
def _set(job_id: str, **fields: Any) -> None:
|
||||
job = get_job(job_id)
|
||||
if job is None:
|
||||
return
|
||||
job.update(fields)
|
||||
try:
|
||||
_write(job_id, job)
|
||||
except OSError:
|
||||
logger.warning("insight_jobs write failed for %s", job_id, exc_info=True)
|
||||
|
||||
|
||||
def _run(job_id: str, params: dict[str, Any]) -> None:
|
||||
from apps.operations.services.insight_service import generate_insight
|
||||
|
||||
def stage_cb(stage: str) -> None:
|
||||
_set(job_id, status="running", stage=stage)
|
||||
|
||||
_set(job_id, status="running", stage="starting")
|
||||
try:
|
||||
result = generate_insight(**params, stage_cb=stage_cb)
|
||||
except Exception as exc: # noqa: BLE001 - surfaced to client via job status
|
||||
logger.exception("insight job %s failed", job_id)
|
||||
_set(job_id, status="error", stage="error", error=str(exc))
|
||||
return
|
||||
_set(job_id, status="done", stage="done", result=result)
|
||||
@@ -10,7 +10,11 @@ from __future__ import annotations
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
from typing import Any
|
||||
import threading
|
||||
import time
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable
|
||||
|
||||
import httpx
|
||||
from django.conf import settings
|
||||
@@ -23,6 +27,9 @@ from apps.operations.services import root_cause
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Plan §5: one LLM inference at a time (CPU server; prevents thread thrash).
|
||||
_INFERENCE_LOCK = threading.Lock()
|
||||
|
||||
_NUMERIC_PIPE_ROW = re.compile(r"^[\d.,]+(\s*\|\s*[\d.,]*)+$")
|
||||
|
||||
# Stored in AIInsight.alert — condition of graded data, not narrative text.
|
||||
@@ -434,7 +441,7 @@ def fetch_rag_chunks(query: str, topic: str, n_results: int = 4) -> tuple[list[s
|
||||
return [], []
|
||||
|
||||
|
||||
def call_ollama(system_prompt: str, user_prompt: str) -> str | None:
|
||||
def call_ollama(system_prompt: str, user_prompt: str) -> tuple[str | None, dict[str, Any]]:
|
||||
model = getattr(settings, "LLM_MODEL_NAME", "qwen2.5:3b") or "qwen2.5:3b"
|
||||
base = getattr(settings, "OLLAMA_BASE_URL", "http://127.0.0.1:11434") or "http://127.0.0.1:11434"
|
||||
url = f"{base.rstrip('/')}/api/chat"
|
||||
@@ -442,7 +449,15 @@ def call_ollama(system_prompt: str, user_prompt: str) -> str | None:
|
||||
payload = {
|
||||
"model": model,
|
||||
"stream": False,
|
||||
"options": {"temperature": 0.2},
|
||||
# format:"json" (grammar) keeps the model from rambling in prose; the schema
|
||||
# reminder appended AFTER the page-data JSON keeps it from copying that JSON.
|
||||
"format": "json",
|
||||
"options": {
|
||||
"temperature": float(getattr(settings, "LLM_TEMPERATURE", 0.2) or 0.2),
|
||||
"seed": int(getattr(settings, "LLM_SEED", 42) or 42),
|
||||
"num_ctx": int(getattr(settings, "LLM_NUM_CTX", 2048) or 2048),
|
||||
"num_thread": int(getattr(settings, "LLM_NUM_THREAD", 4) or 4),
|
||||
},
|
||||
"messages": [
|
||||
{"role": "system", "content": system_prompt},
|
||||
{"role": "user", "content": user_prompt},
|
||||
@@ -450,16 +465,23 @@ def call_ollama(system_prompt: str, user_prompt: str) -> str | None:
|
||||
}
|
||||
try:
|
||||
with httpx.Client(timeout=timeout) as client:
|
||||
resp = client.post(url, json=payload)
|
||||
# Plan §5: serialize LLM calls so concurrent generates don't thrash CPU.
|
||||
with _INFERENCE_LOCK:
|
||||
resp = client.post(url, json=payload)
|
||||
if resp.status_code >= 400:
|
||||
logger.error("Ollama HTTP %s: %s", resp.status_code, resp.text[:500])
|
||||
return None
|
||||
return None, {}
|
||||
data = resp.json()
|
||||
message = data.get("message") or {}
|
||||
return message.get("content") or data.get("response")
|
||||
content = message.get("content") or data.get("response")
|
||||
usage = {
|
||||
"prompt_tokens": data.get("prompt_eval_count"),
|
||||
"completion_tokens": data.get("eval_count"),
|
||||
}
|
||||
return content, usage
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.error("Ollama call failed: %s", exc)
|
||||
return None
|
||||
return None, {}
|
||||
|
||||
|
||||
def parse_llm_json(text: str) -> dict[str, str] | None:
|
||||
@@ -530,23 +552,6 @@ def parse_end_cycle_llm_json(text: str) -> dict[str, Any] | None:
|
||||
return parsed
|
||||
|
||||
|
||||
def local_fallback_insight(graded: dict[str, Any], topic: str) -> dict[str, str]:
|
||||
analyses = graded.get("analyses") or {}
|
||||
lines = []
|
||||
for name, block in analyses.items():
|
||||
if isinstance(block, dict) and block.get("message"):
|
||||
lines.append(f"{name}: {block['message']}")
|
||||
if not lines:
|
||||
return {
|
||||
"kesimpulan": f"Data untuk topik {topic} tidak cukup untuk dianalisis (status unknown).",
|
||||
"insight": "Lengkapi data operasional kandang, lalu generate ulang. Jangan mengarang angka.",
|
||||
}
|
||||
return {
|
||||
"kesimpulan": lines[0],
|
||||
"insight": "\n".join(lines[1:]) if len(lines) > 1 else lines[0],
|
||||
}
|
||||
|
||||
|
||||
_SOFT_MORTALITY_WORD = re.compile(
|
||||
r"\b(rendah|normal|baik|aman|terkendali|sedikit)\b",
|
||||
re.IGNORECASE,
|
||||
@@ -616,6 +621,11 @@ def collapse_duplicate_kandang(text: str) -> str:
|
||||
return _DUP_KANDANG.sub(lambda m: "Kandang" if m.group(0)[0].isupper() else "kandang", text)
|
||||
|
||||
|
||||
def _placeholder_text(value: str) -> bool:
|
||||
"""True when a narrative field is empty / schema placeholder (e.g. '...')."""
|
||||
return len(value.strip().strip(".…<>- \t")) < 20
|
||||
|
||||
|
||||
def sanitize_insight_narrative(parsed: dict[str, Any]) -> dict[str, Any]:
|
||||
"""Deterministic cleanup of narrative fields after the LLM."""
|
||||
out = dict(parsed)
|
||||
@@ -736,6 +746,50 @@ def serialize_insight(row: AIInsight) -> dict[str, Any]:
|
||||
}
|
||||
|
||||
|
||||
# Plan Stage 1: graded status -> deterministic RAG keywords (no LLM query rewrite).
|
||||
_DIAGNOSTIC_KEYWORDS = {
|
||||
"mortality": "penanganan mortalitas kumulatif deplesi penyebab kematian brooding",
|
||||
"fcr": "target FCR pakan harian konsumsi pakan efisiensi ventilasi",
|
||||
"bw": "target bobot badan ADG pertumbuhan standar mingguan",
|
||||
"iot": "standar ventilasi suhu kelembapan amonia sekam basah kandang",
|
||||
"environment": "standar ventilasi suhu kelembapan amonia sekam basah kandang",
|
||||
}
|
||||
|
||||
|
||||
def diagnose_rag_keywords(graded: dict[str, Any]) -> str:
|
||||
"""Stage 1: anomaly keywords from graded blocks; '' when everything ok/unknown."""
|
||||
analyses = (graded or {}).get("analyses") or {}
|
||||
clues: list[str] = []
|
||||
for key, block in analyses.items():
|
||||
if not isinstance(block, dict):
|
||||
continue
|
||||
if str(block.get("status") or "").lower() in ("warning", "critical"):
|
||||
hint = _DIAGNOSTIC_KEYWORDS.get(key)
|
||||
if hint and hint not in clues:
|
||||
clues.append(hint)
|
||||
return " ".join(clues)
|
||||
|
||||
|
||||
def _notify(stage_cb: Callable[[str], None] | None, stage: str) -> None:
|
||||
if not stage_cb:
|
||||
return
|
||||
try:
|
||||
stage_cb(stage)
|
||||
except Exception: # noqa: BLE001 - progress UI must never break generation
|
||||
logger.debug("stage_cb(%s) failed", stage, exc_info=True)
|
||||
|
||||
|
||||
def _log_trace(record: dict[str, Any]) -> None:
|
||||
"""Plan Stage 5: append-only audit trail for replay/debug."""
|
||||
try:
|
||||
path = Path(settings.BASE_DIR) / "logs" / "rag_trace.jsonl"
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with path.open("a", encoding="utf-8") as fh:
|
||||
fh.write(json.dumps(record, ensure_ascii=False, default=str) + "\n")
|
||||
except Exception: # noqa: BLE001 - logging must never fail the request
|
||||
logger.warning("rag_trace write failed", exc_info=True)
|
||||
|
||||
|
||||
def generate_insight(
|
||||
*,
|
||||
cycle_id: int,
|
||||
@@ -745,6 +799,7 @@ def generate_insight(
|
||||
report_type: str = "page",
|
||||
report_period: str = "current",
|
||||
force_refresh: bool = False,
|
||||
stage_cb: Callable[[str], None] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
if not kandang_id:
|
||||
raise ValueError("kandang_id wajib — insight tidak boleh untuk semua kandang")
|
||||
@@ -785,7 +840,11 @@ def generate_insight(
|
||||
ctx.setdefault("cycleId", cycle_id)
|
||||
ctx = prune_context_by_period(ctx, report_type, report_period)
|
||||
|
||||
t_start = time.monotonic()
|
||||
_notify(stage_cb, "grading")
|
||||
graded = grade_context(ctx)
|
||||
t_after_grade = time.monotonic()
|
||||
_notify(stage_cb, "retrieving")
|
||||
day = graded.get("hari_ke")
|
||||
standard_block = cp707.build_cp707_standard_block(ctx) or cp707.build_book_reference_context(day or 1)
|
||||
|
||||
@@ -805,7 +864,10 @@ def generate_insight(
|
||||
)
|
||||
else:
|
||||
rag_query = f"panduan manajemen broiler CP 707 untuk {topic} umur hari ke-{day or '?'}"
|
||||
rag_query = f"{rag_query} {diagnose_rag_keywords(graded)}".strip()
|
||||
chunks, citations = fetch_rag_chunks(rag_query, topic)
|
||||
t_after_rag = time.monotonic()
|
||||
_notify(stage_cb, "synthesizing")
|
||||
|
||||
system_prompt = (
|
||||
"Anda adalah asisten farm broiler on-premise. Tugas Anda HANYA menulis narasi "
|
||||
@@ -829,8 +891,12 @@ def generate_insight(
|
||||
f"\n[NEXT CYCLE ACTIONS]\n"
|
||||
f"{json.dumps(next_actions, ensure_ascii=False, default=str)}\n"
|
||||
)
|
||||
cited_chunks = [
|
||||
f"[cp707#chunk{(citations[i].get('chunk_id') if i < len(citations) else None) or i}] {chunk}"
|
||||
for i, chunk in enumerate(chunks[:4])
|
||||
]
|
||||
if chunks:
|
||||
system_prompt += "\n[CUPLIKAN SOP CP 707 — prosa]\n" + "\n---\n".join(chunks[:4])
|
||||
system_prompt += "\n[CUPLIKAN SOP CP 707 — prosa]\n" + "\n---\n".join(cited_chunks)
|
||||
|
||||
user_prompt = build_insight_user_prompt(
|
||||
topic=topic,
|
||||
@@ -839,44 +905,152 @@ def generate_insight(
|
||||
report_period=report_period,
|
||||
context=ctx,
|
||||
)
|
||||
# Page-data JSON ends the user message; small models copy the last JSON they
|
||||
# read. Re-state the required schema AFTER the data so recency wins.
|
||||
# Status/facts go right before it: book chunks (system tail) describe IDEAL
|
||||
# conditions, so without this last the model narrates ideals over actual status.
|
||||
# Skip "unknown" blocks when at least one block has a real status:
|
||||
# no-data lines are padding triggers (e.g. mortality spun into FCR
|
||||
# narrative). All-unknown case keeps them: nothing factual to lean on
|
||||
# and "data tidak tersedia" is then the only correct narration.
|
||||
known_lines = [
|
||||
f"{name} = {block.get('status')}: {block.get('message')}"
|
||||
for name, block in (graded.get("analyses") or {}).items()
|
||||
if isinstance(block, dict)
|
||||
and block.get("message")
|
||||
and block.get("status") != "unknown"
|
||||
]
|
||||
unknown_lines = [
|
||||
f"{name} = {block.get('status')}: {block.get('message')}"
|
||||
for name, block in (graded.get("analyses") or {}).items()
|
||||
if isinstance(block, dict)
|
||||
and block.get("message")
|
||||
and block.get("status") == "unknown"
|
||||
]
|
||||
status_lines = known_lines or unknown_lines
|
||||
actual_alert = alert_from_graded(graded)
|
||||
user_prompt = (user_prompt or "").rstrip() + (
|
||||
"\nKONTEKS: semua data adalah ternak AYAM BROILER (unggas) di kandang — "
|
||||
"BUKAN tanaman/pertanian. 'Panen' = panen ayam. 'Bobot' = gram per ekor "
|
||||
"(bukan per karung). 'ADG' = kenaikan bobot harian ayam."
|
||||
f"\nSTATUS AKTUAL: {actual_alert}. Kesimpulan dan insight HARUS konsisten "
|
||||
"dengan status ini: bila warning/critical, sebut masalahnya dan bandingkan "
|
||||
"dengan angka standar CP 707; dilarang menyebut ideal/sehat/baik/aman "
|
||||
"bila STATUS AKTUAL bukan healthy."
|
||||
+ ("\nFAKTA GRADED:\n- " + "\n- ".join(status_lines) if status_lines else "")
|
||||
# No length req anywhere -> 3B model sometimes stops after 1 sentence
|
||||
# (three EEF rows stored identical 136/127 chars). Demand structure.
|
||||
+ "\nPANJANG WAJIB: insight = narasi 3-6 kalimat berurutan: "
|
||||
"(1) angka aktual vs standar CP 707, (2) penyebab/implikasi, "
|
||||
"(3) tindakan konkret. kesimpulan = 1-3 kalimat. "
|
||||
"Satu kalimat singkat = gagal."
|
||||
# Placeholder <> (not "..."): model has copied schema "..." verbatim
|
||||
# into insight, which stored fine and rendered as an empty page.
|
||||
+ '\nBALAS HANYA JSON persis: {"kesimpulan":"<ringkasan>","insight":"<narasi panjang>"}'
|
||||
+ ' - JANGAN salin teks <> dari contoh; tulis kalimat lengkap sendiri.'
|
||||
)
|
||||
|
||||
raw = call_ollama(system_prompt, user_prompt)
|
||||
if is_end_cycle:
|
||||
parsed_raw = parse_end_cycle_llm_json(raw or "")
|
||||
if parsed_raw:
|
||||
# Apply mortality wording on narrative fields before sanitize.
|
||||
soft = {
|
||||
"kesimpulan": str(parsed_raw.get("kesimpulan") or ""),
|
||||
"insight": str(parsed_raw.get("insight") or ""),
|
||||
}
|
||||
soft = enforce_mortality_wording(soft, graded)
|
||||
soft = sanitize_insight_narrative(soft)
|
||||
parsed_raw["kesimpulan"] = soft.get("kesimpulan")
|
||||
parsed_raw["insight"] = soft.get("insight")
|
||||
payload = root_cause.sanitize_end_cycle_payload(
|
||||
parsed_raw,
|
||||
hypotheses=hypotheses,
|
||||
actions=next_actions,
|
||||
masalah=masalah,
|
||||
# Re-state output schema as the system prompt's final line (user_prompt
|
||||
# repeats it again after the page-data JSON).
|
||||
system_prompt += (
|
||||
"\nINGAT: keluaran HANYA JSON valid sesuai FORMAT OUTPUT di atas "
|
||||
"(wajib ada key kesimpulan dan insight). Tanpa teks lain."
|
||||
)
|
||||
# Degenerate output (schema placeholder like "..." copied verbatim) stored
|
||||
# fine as JSON and rendered as an empty insight page. Reject + retry once,
|
||||
# then fail loud (same policy as parse failure).
|
||||
degenerate_reason: str | None = None
|
||||
for attempt in range(2):
|
||||
raw, usage = call_ollama(system_prompt, user_prompt)
|
||||
t_after_llm = time.monotonic()
|
||||
if is_end_cycle:
|
||||
parsed_raw = parse_end_cycle_llm_json(raw or "")
|
||||
if parsed_raw:
|
||||
# Apply mortality wording on narrative fields before sanitize.
|
||||
soft = {
|
||||
"kesimpulan": str(parsed_raw.get("kesimpulan") or ""),
|
||||
"insight": str(parsed_raw.get("insight") or ""),
|
||||
}
|
||||
soft = enforce_mortality_wording(soft, graded)
|
||||
soft = sanitize_insight_narrative(soft)
|
||||
parsed_raw["kesimpulan"] = soft.get("kesimpulan")
|
||||
parsed_raw["insight"] = soft.get("insight")
|
||||
payload = root_cause.sanitize_end_cycle_payload(
|
||||
parsed_raw,
|
||||
hypotheses=hypotheses,
|
||||
actions=next_actions,
|
||||
masalah=masalah,
|
||||
)
|
||||
payload = sanitize_insight_narrative(payload)
|
||||
else:
|
||||
# Fallback removed (2026-09-25): fail loudly instead of serving fake insight.
|
||||
if not raw:
|
||||
raise RuntimeError("Ollama tidak mengembalikan output apa pun (cek layanan LLM).")
|
||||
raise RuntimeError(
|
||||
"Output LLM end-cycle tidak sesuai format JSON yang diminta: "
|
||||
+ repr(str(raw)[:200])
|
||||
)
|
||||
summary = collapse_duplicate_kandang(str(payload.get("kesimpulan") or ""))
|
||||
insight_text = collapse_duplicate_kandang(
|
||||
root_cause.format_end_cycle_insight_text(payload)
|
||||
)
|
||||
payload = sanitize_insight_narrative(payload)
|
||||
else:
|
||||
payload = root_cause.local_fallback_end_cycle(graded, ctx)
|
||||
summary = collapse_duplicate_kandang(str(payload.get("kesimpulan") or ""))
|
||||
insight_text = collapse_duplicate_kandang(
|
||||
root_cause.format_end_cycle_insight_text(payload)
|
||||
)
|
||||
else:
|
||||
parsed = parse_llm_json(raw or "")
|
||||
if not parsed:
|
||||
parsed = local_fallback_insight(graded, topic)
|
||||
else:
|
||||
parsed = parse_llm_json(raw or "")
|
||||
if not parsed:
|
||||
# Fallback removed (2026-09-25): fail loudly instead of serving fake insight.
|
||||
if not raw:
|
||||
raise RuntimeError("Ollama tidak mengembalikan output apa pun (cek layanan LLM).")
|
||||
raise RuntimeError(
|
||||
"Output LLM tidak sesuai format JSON yang diminta: " + repr(str(raw)[:200])
|
||||
)
|
||||
parsed = enforce_mortality_wording(parsed, graded)
|
||||
parsed = sanitize_insight_narrative(parsed)
|
||||
insight_text = parsed["insight"]
|
||||
summary = parsed["kesimpulan"]
|
||||
parsed = sanitize_insight_narrative(parsed)
|
||||
insight_text = parsed["insight"]
|
||||
summary = parsed["kesimpulan"]
|
||||
if _placeholder_text(summary) or _placeholder_text(insight_text):
|
||||
degenerate_reason = f"kesimpulan={summary!r} insight={insight_text!r}"
|
||||
user_prompt += (
|
||||
"\nCATATAN: keluaran sebelumnya hanya placeholder. "
|
||||
"Tulis kalimat lengkap sendiri untuk kesimpulan dan insight."
|
||||
)
|
||||
continue
|
||||
degenerate_reason = None
|
||||
break
|
||||
if degenerate_reason:
|
||||
raise RuntimeError(
|
||||
"Output LLM degenerate (placeholder bukan kalimat): "
|
||||
+ degenerate_reason[:200]
|
||||
)
|
||||
|
||||
alert = alert_from_graded(graded)
|
||||
_notify(stage_cb, "saving")
|
||||
_log_trace(
|
||||
{
|
||||
"timestamp": timezone.now().isoformat(),
|
||||
"request_id": str(uuid.uuid4()),
|
||||
"cycle_id": cycle_id,
|
||||
"kandang_name": kandang.kandang_name,
|
||||
"topic": topic,
|
||||
"report_type": report_type,
|
||||
"report_period": report_period,
|
||||
"model": getattr(settings, "LLM_MODEL_NAME", None),
|
||||
"prompt_version": "v2.1",
|
||||
"formulated_query": rag_query,
|
||||
"retrieved_chunk_ids": [
|
||||
f"cp707#{c.get('chunk_id')}" if c.get("chunk_id") is not None else f"cp707#{c.get('source')}"
|
||||
for c in citations
|
||||
],
|
||||
"graded_alert": alert,
|
||||
"llm_ok": raw is not None,
|
||||
"latency_breakdown": {
|
||||
"grade_ms": round((t_after_grade - t_start) * 1000),
|
||||
"retrieval_ms": round((t_after_rag - t_after_grade) * 1000),
|
||||
"llm_ms": round((t_after_llm - t_after_rag) * 1000),
|
||||
"total_ms": round((time.monotonic() - t_start) * 1000),
|
||||
},
|
||||
"token_usage": usage,
|
||||
}
|
||||
)
|
||||
|
||||
row, _created = AIInsight.objects.update_or_create(
|
||||
cycle=cycle,
|
||||
|
||||
@@ -323,6 +323,25 @@ class AIInsightViewSet(CycleScopedViewSet):
|
||||
{"detail": "context.kandangId must match kandang_id."},
|
||||
status=status.HTTP_400_BAD_REQUEST,
|
||||
)
|
||||
# Phase 4: async mode returns 202 + job_id (client polls jobs/{id}/).
|
||||
if bool(data.get("async")):
|
||||
from apps.operations.services import insight_jobs
|
||||
|
||||
job_id = insight_jobs.create_job(
|
||||
{
|
||||
"cycle_id": int(cycle_id),
|
||||
"kandang_id": int(kandang_id),
|
||||
"topic": str(topic),
|
||||
"context": context if isinstance(context, dict) else {},
|
||||
"report_type": str(report_type),
|
||||
"report_period": str(report_period),
|
||||
"force_refresh": force_refresh,
|
||||
}
|
||||
)
|
||||
return Response(
|
||||
{"job_id": job_id, "status": "queued"},
|
||||
status=status.HTTP_202_ACCEPTED,
|
||||
)
|
||||
try:
|
||||
result = generate_insight(
|
||||
cycle_id=int(cycle_id),
|
||||
@@ -342,6 +361,15 @@ class AIInsightViewSet(CycleScopedViewSet):
|
||||
)
|
||||
return Response(result)
|
||||
|
||||
@action(detail=False, methods=["get"], url_path=r"jobs/(?P<job_id>[^/.]+)")
|
||||
def job_status(self, request, job_id: str | None = None):
|
||||
from apps.operations.services import insight_jobs
|
||||
|
||||
job = insight_jobs.get_job(job_id or "")
|
||||
if job is None:
|
||||
return Response({"detail": "Job not found."}, status=status.HTTP_404_NOT_FOUND)
|
||||
return Response(job)
|
||||
|
||||
@action(detail=False, methods=["get"], url_path="cached")
|
||||
def cached(self, request):
|
||||
from apps.operations.services.insight_service import get_cached_insight
|
||||
|
||||
Reference in new issue
Block a user