fix and adjust main dashboard ai insight

This commit is contained in:
Alberto-Audrix committed 2026-09-17 14:27:08 +07:00
1 parent 18790ab8f6
commit 68e0d1a0bb
27 files changed
+360 -107

No files matched your search

+34
View File
@@ -0,0 +1,34 @@
"""Classify CP 707 extracted chunks as prosa vs tabel for RAG metadata."""
from __future__ import annotations
import re
_NUMERIC_PIPE_ROW = re.compile(r"^[\d.,]+(\s*\|\s*[\d.,]*)+$")
_CHAPTER_RE = re.compile(
r"(?:bab|chapter|lampiran)\s*([0-9IVXLC]+)",
re.IGNORECASE,
)
def classify_chunk_tipe(text: str) -> str:
"""Return 'tabel' if the chunk is mostly headerless numeric rows, else 'prosa'."""
lines = [ln.strip() for ln in (text or "").splitlines() if ln.strip()]
if not lines:
return "prosa"
numeric = sum(1 for ln in lines if _NUMERIC_PIPE_ROW.match(ln))
if numeric >= max(2, len(lines) // 2):
return "tabel"
return "prosa"
def detect_bab(text: str, source_filename: str = "") -> str:
"""Best-effort chapter / lampiran label from chunk text or filename."""
haystack = f"{source_filename}\n{text[:800]}"
match = _CHAPTER_RE.search(haystack)
if match:
return match.group(0).strip()
lower = source_filename.lower()
if "lampiran" in lower:
return "lampiran"
return ""