"""Classify CP 707 extracted chunks as prosa vs tabel for RAG metadata.""" from __future__ import annotations import re _NUMERIC_PIPE_ROW = re.compile(r"^[\d.,]+(\s*\|\s*[\d.,]*)+$") _CHAPTER_RE = re.compile( r"(?:bab|chapter|lampiran)\s*([0-9IVXLC]+)", re.IGNORECASE, ) def classify_chunk_tipe(text: str) -> str: """Return 'tabel' if the chunk is mostly headerless numeric rows, else 'prosa'.""" lines = [ln.strip() for ln in (text or "").splitlines() if ln.strip()] if not lines: return "prosa" numeric = sum(1 for ln in lines if _NUMERIC_PIPE_ROW.match(ln)) if numeric >= max(2, len(lines) // 2): return "tabel" return "prosa" def detect_bab(text: str, source_filename: str = "") -> str: """Best-effort chapter / lampiran label from chunk text or filename.""" haystack = f"{source_filename}\n{text[:800]}" match = _CHAPTER_RE.search(haystack) if match: return match.group(0).strip() lower = source_filename.lower() if "lampiran" in lower: return "lampiran" return ""