35 lines
1.0 KiB
Python
35 lines
1.0 KiB
Python
"""Classify CP 707 extracted chunks as prosa vs tabel for RAG metadata."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
|
|
_NUMERIC_PIPE_ROW = re.compile(r"^[\d.,]+(\s*\|\s*[\d.,]*)+$")
|
|
_CHAPTER_RE = re.compile(
|
|
r"(?:bab|chapter|lampiran)\s*([0-9IVXLC]+)",
|
|
re.IGNORECASE,
|
|
)
|
|
|
|
|
|
def classify_chunk_tipe(text: str) -> str:
|
|
"""Return 'tabel' if the chunk is mostly headerless numeric rows, else 'prosa'."""
|
|
lines = [ln.strip() for ln in (text or "").splitlines() if ln.strip()]
|
|
if not lines:
|
|
return "prosa"
|
|
numeric = sum(1 for ln in lines if _NUMERIC_PIPE_ROW.match(ln))
|
|
if numeric >= max(2, len(lines) // 2):
|
|
return "tabel"
|
|
return "prosa"
|
|
|
|
|
|
def detect_bab(text: str, source_filename: str = "") -> str:
|
|
"""Best-effort chapter / lampiran label from chunk text or filename."""
|
|
haystack = f"{source_filename}\n{text[:800]}"
|
|
match = _CHAPTER_RE.search(haystack)
|
|
if match:
|
|
return match.group(0).strip()
|
|
lower = source_filename.lower()
|
|
if "lampiran" in lower:
|
|
return "lampiran"
|
|
return ""
|