40 lines
1.1 KiB
Python
40 lines
1.1 KiB
Python
"""Classify CP 707 extracted chunks as prosa vs tabel for RAG metadata."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
|
|
_NUMERIC_PIPE_ROW = re.compile(r"^[\d.,]+(\s*\|\s*[\d.,]*)+$")
|
|
_CHAPTER_RE = re.compile(
|
|
r"(?:bab|chapter|lampiran)\s*([0-9IVXLC]+)",
|
|
re.IGNORECASE,
|
|
)
|
|
|
|
|
|
def classify_chunk_tipe(text: str) -> str:
|
|
"""Return 'tabel' if the chunk is mostly headerless numeric rows, else 'prosa'."""
|
|
lines = [ln.strip() for ln in (text or "").splitlines() if ln.strip()]
|
|
if not lines:
|
|
return "prosa"
|
|
numeric = sum(
|
|
1
|
|
for ln in lines
|
|
if _NUMERIC_PIPE_ROW.match(ln)
|
|
or (ln.startswith("|") and ln.endswith("|") and ln.count("|") > 1)
|
|
)
|
|
if numeric >= max(2, len(lines) // 2):
|
|
return "tabel"
|
|
return "prosa"
|
|
|
|
|
|
def detect_bab(text: str, source_filename: str = "") -> str:
|
|
"""Best-effort chapter / lampiran label from chunk text or filename."""
|
|
haystack = f"{source_filename}\n{text[:800]}"
|
|
match = _CHAPTER_RE.search(haystack)
|
|
if match:
|
|
return match.group(0).strip()
|
|
lower = source_filename.lower()
|
|
if "lampiran" in lower:
|
|
return "lampiran"
|
|
return ""
|