Files
dashboard-cpsp/ai-insight/chunk_classify.py
T
2026-09-29 15:19:10 +07:00

40 lines
1.1 KiB
Python

"""Classify CP 707 extracted chunks as prosa vs tabel for RAG metadata."""
from __future__ import annotations
import re
_NUMERIC_PIPE_ROW = re.compile(r"^[\d.,]+(\s*\|\s*[\d.,]*)+$")
_CHAPTER_RE = re.compile(
r"(?:bab|chapter|lampiran)\s*([0-9IVXLC]+)",
re.IGNORECASE,
)
def classify_chunk_tipe(text: str) -> str:
"""Return 'tabel' if the chunk is mostly headerless numeric rows, else 'prosa'."""
lines = [ln.strip() for ln in (text or "").splitlines() if ln.strip()]
if not lines:
return "prosa"
numeric = sum(
1
for ln in lines
if _NUMERIC_PIPE_ROW.match(ln)
or (ln.startswith("|") and ln.endswith("|") and ln.count("|") > 1)
)
if numeric >= max(2, len(lines) // 2):
return "tabel"
return "prosa"
def detect_bab(text: str, source_filename: str = "") -> str:
"""Best-effort chapter / lampiran label from chunk text or filename."""
haystack = f"{source_filename}\n{text[:800]}"
match = _CHAPTER_RE.search(haystack)
if match:
return match.group(0).strip()
lower = source_filename.lower()
if "lampiran" in lower:
return "lampiran"
return ""