#!/usr/bin/env python3 """ Publication-Grade Document Compiler for LibreOffice Flat XML (.fodt) and PDF. Milestone 3 - Retraining & Live Counting Documentation """ import os import sys import re import base64 import xml.etree.ElementTree as ET from PIL import Image WORKSPACE_ROOT = "/home/asus/feedmill/reTraining" MD_PATH = os.path.join(WORKSPACE_ROOT, "docs/PANDUAN_SISTEM_LENGKAP.md") FODT_PATH = os.path.join(WORKSPACE_ROOT, "docs/PANDUAN_SISTEM_LENGKAP.fodt") PDF_PATH = os.path.join(WORKSPACE_ROOT, "docs/PANDUAN_SISTEM_LENGKAP.pdf") def escape_xml(s): if s is None: return "" return (s.replace("&", "&") .replace("<", "<") .replace(">", ">") .replace('"', """) .replace("'", "'")) def escape_html(s): if s is None: return "" return (s.replace("&", "&") .replace("<", "<") .replace(">", ">") .replace('"', """)) def get_image_info(rel_path): # Determine full path if rel_path.startswith("docs/"): full_path = os.path.join(WORKSPACE_ROOT, rel_path) elif os.path.exists(os.path.join(WORKSPACE_ROOT, "docs", rel_path)): full_path = os.path.join(WORKSPACE_ROOT, "docs", rel_path) elif os.path.exists(os.path.join(WORKSPACE_ROOT, rel_path)): full_path = os.path.join(WORKSPACE_ROOT, rel_path) else: raise FileNotFoundError(f"Image not found: {rel_path}") with open(full_path, "rb") as f: data = f.read() b64_str = base64.b64encode(data).decode("ascii") with Image.open(full_path) as img: w_px, h_px = img.size # Printable width in A4 is 17.0 cm (21.0 - 2.0 - 2.0) aspect = w_px / h_px if aspect > 1.7: # Diagram (16:9) width_cm = 16.5 height_cm = width_cm / aspect else: # 16:10 screenshots width_cm = 15.8 height_cm = width_cm / aspect return { "full_path": full_path, "base64": b64_str, "width_px": w_px, "height_px": h_px, "width_cm": round(width_cm, 2), "height_cm": round(height_cm, 2), "data_uri": f"data:image/png;base64,{b64_str}" } def parse_inlines_to_odf(text): # Tokenize inline formatting: code, bolditalic, bold, italic, links, math pattern = re.compile(r"(`[^`]+`|\*\*\*[^*]+\*\*\*|\*\*[^*]+\*\*|\*[^*]+\*|\[[^\]]+\]\([^)]+\)|\$[^$]+\$)") pos = 0 result = [] for m in pattern.finditer(text): if m.start() > pos: result.append(escape_xml(text[pos:m.start()])) token = m.group(0) if token.startswith("`") and token.endswith("`"): code_text = escape_xml(token[1:-1]) result.append(f'{code_text}') elif token.startswith("***") and token.endswith("***"): t = escape_xml(token[3:-3]) result.append(f'{t}') elif token.startswith("**") and token.endswith("**"): t = escape_xml(token[2:-2]) result.append(f'{t}') elif token.startswith("*") and token.endswith("*"): t = escape_xml(token[1:-1]) result.append(f'{t}') elif token.startswith("[") and "]" in token and "(" in token: lm = re.match(r"\[([^\]]+)\]\(([^)]+)\)", token) if lm: lt, lu = lm.groups() result.append(f'{escape_xml(lt)}') else: result.append(escape_xml(token)) elif token.startswith("$") and token.endswith("$"): t = escape_xml(token[1:-1]) result.append(f'{t}') else: result.append(escape_xml(token)) pos = m.end() if pos < len(text): result.append(escape_xml(text[pos:])) return "".join(result) def parse_inlines_to_html(text): text = escape_html(text) # Inline code text = re.sub(r"`([^`]+)`", r'\1', text) # Bold italic text = re.sub(r"\*\*\*([^*]+)\*\*\*", r"\1", text) # Bold text = re.sub(r"\*\*([^*]+)\*\*", r"\1", text) # Italic text = re.sub(r"\*([^*]+)\*", r"\1", text) # Links text = re.sub(r"\[([^\]]+)\]\(([^)]+)\)", r'\1', text) # Math var text = re.sub(r"\$([^$]+)\$", r'\1', text) return text def parse_markdown_blocks(content): lines = content.split("\n") blocks = [] i = 0 n = len(lines) while i < n: line = lines[i] stripped = line.strip() # Empty line if not stripped: i += 1 continue # Horizontal Rule if stripped in ["---", "***", "___"]: blocks.append({"type": "hr"}) i += 1 continue # Code block if stripped.startswith("```"): lang = stripped[3:].strip() code_lines = [] i += 1 while i < n and not lines[i].strip().startswith("```"): code_lines.append(lines[i]) i += 1 if i < n: # consume closing ``` i += 1 blocks.append({"type": "code", "lang": lang, "content": "\n".join(code_lines)}) continue # Headings if stripped.startswith("#"): level = 0 while level < len(stripped) and stripped[level] == "#": level += 1 title = stripped[level:].strip() blocks.append({"type": "heading", "level": level, "text": title}) i += 1 continue # Table if stripped.startswith("|") and stripped.endswith("|"): table_lines = [] while i < n and lines[i].strip().startswith("|") and lines[i].strip().endswith("|"): table_lines.append(lines[i].strip()) i += 1 if len(table_lines) >= 2: header = [c.strip() for c in table_lines[0].split("|")[1:-1]] rows = [] for r in table_lines[2:]: cols = [c.strip() for c in r.split("|")[1:-1]] rows.append(cols) blocks.append({"type": "table", "header": header, "rows": rows}) continue # Image img_match = re.match(r"^!\[(.*?)\]\((.*?)\)$", stripped) if img_match: alt, src = img_match.groups() blocks.append({"type": "image", "alt": alt, "src": src}) i += 1 continue # Numbered list: 1. num_match = re.match(r"^(\d+)\.\s+(.*)$", stripped) if num_match: list_items = [] while i < n: curr_stripped = lines[i].strip() nm = re.match(r"^(\d+)\.\s+(.*)$", curr_stripped) if nm: list_items.append(nm.group(2)) i += 1 else: break blocks.append({"type": "ordered_list", "items": list_items}) continue # Bullet list: - or * bullet_match = re.match(r"^[-*]\s+(.*)$", stripped) if bullet_match and not re.match(r"^\*[^*]+\*$", stripped): bullet_items = [] while i < n: curr_stripped = lines[i].strip() bm = re.match(r"^[-*]\s+(.*)$", curr_stripped) if bm and not re.match(r"^\*[^*]+\*$", curr_stripped): bullet_items.append(bm.group(1)) i += 1 else: break blocks.append({"type": "unordered_list", "items": bullet_items}) continue # Callout / Admonition (*Perhatian:...* or *Invarian penting:...*) if stripped.startswith("*Perhatian:") or stripped.startswith("*Invarian penting:") or stripped.startswith("*Catatan:") or stripped.startswith("*Peringatan:"): blocks.append({"type": "admonition", "text": stripped}) i += 1 continue # Caption or italic paragraph under image (*Gambar X:...* or *Unduh format...*) if stripped.startswith("*Gambar ") or stripped.startswith("*Unduh format vektor"): blocks.append({"type": "caption", "text": stripped}) i += 1 continue # Standard paragraph p_lines = [stripped] i += 1 while i < n: next_line = lines[i] next_stripped = next_line.strip() if (not next_stripped or next_stripped.startswith("#") or next_stripped.startswith("```") or next_stripped.startswith("|") or next_stripped.startswith("---") or next_stripped.startswith("![") or re.match(r"^\d+\.\s+", next_stripped) or re.match(r"^[-*]\s+", next_stripped) or next_stripped.startswith("*Perhatian:") or next_stripped.startswith("*Invarian penting:") or next_stripped.startswith("*Gambar ") or next_stripped.startswith("*Unduh format")): break p_lines.append(next_stripped) i += 1 blocks.append({"type": "paragraph", "text": " ".join(p_lines)}) return blocks print("Parser module loaded successfully.") def generate_fodt(blocks): print("Generating OASIS OpenDocument Text Flat XML (.fodt)...") # Header out = [] out.append('') out.append('') # Meta out.append(' ') out.append(' PANDUAN SISTEM LENGKAP: RETRAINING, ANOTASI & LIVE COUNTING KARUNG KONVEYOR') out.append(' Dokumentasi Arsitektur, Prosedur Operasional Standar, dan Panduan Referensi Teknis Produksi v4.2.0') out.append(' id-ID') out.append(' Senior Computer Vision & Platform Engineer') out.append(' 2026-08-27T00:00:00Z') out.append(' ') # Font face declarations out.append(' ') out.append(' ') out.append(' ') out.append(' ') # Styles out.append(' ') # Default graphic out.append(' ') out.append(' ') out.append(' ') # Default paragraph out.append(' ') out.append(' ') out.append(' ') out.append(' ') # Default table out.append(' ') out.append(' ') out.append(' ') # Default table cell out.append(' ') out.append(' ') out.append(' ') # Named Styles out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') # List styles out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') # Automatic styles out.append(' ') # Page layout A4 (21.0 x 29.7 cm, 2.0 cm margins) out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') # Character styles out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') # Graphic Frame style out.append(' ') out.append(' ') out.append(' ') # Table styles out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append(' ') # Master styles out.append(' ') out.append(' ') out.append(' ') out.append(' PANDUAN SISTEM LENGKAP — RETRAINING, ANOTASI & LIVE COUNTING') out.append(' ') out.append(' ') out.append(' Dokumentasi Teknis Produksi v4.2.0 | Halaman 1 dari 1') out.append(' ') out.append(' ') out.append(' ') # Document Body out.append(' ') out.append(' ') img_idx = 0 tbl_idx = 0 first_h1 = True for block in blocks: btype = block["type"] if btype == "hr": # Horizontal separator out.append(' ') elif btype == "heading": level = block["level"] text = block["text"] if level == 1: # Check if it is Document Title or Chapter if first_h1 and "PANDUAN SISTEM LENGKAP" in text: out.append(f' {parse_inlines_to_odf(text)}') first_h1 = False else: out.append(f' {parse_inlines_to_odf(text)}') elif level == 2: out.append(f' {parse_inlines_to_odf(text)}') elif level == 3: out.append(f' {parse_inlines_to_odf(text)}') else: out.append(f' {parse_inlines_to_odf(text)}') elif btype == "paragraph": text = block["text"] if text.startswith("**Dokumentasi Arsitektur"): out.append(f' {parse_inlines_to_odf(text)}') else: out.append(f' {parse_inlines_to_odf(text)}') elif btype == "admonition": text = block["text"] style = "AdmonitionWarning" if "Perhatian" in text or "Peringatan" in text else "Admonition" out.append(f' {parse_inlines_to_odf(text)}') elif btype == "code": code_text = block["content"] lines = code_text.split("\n") while lines and not lines[-1].strip(): lines.pop() xml_lines = [] for cline in lines: escaped = escape_xml(cline) def replace_spaces(match): count = len(match.group(0)) return f'' processed_line = re.sub(r" {2,}", replace_spaces, escaped) xml_lines.append(processed_line) inner = "".join(xml_lines) out.append(f' {inner}') elif btype == "image": img_idx += 1 src = block["src"] alt = block["alt"] img_info = get_image_info(src) w_cm = img_info["width_cm"] h_cm = img_info["height_cm"] b64 = img_info["base64"] out.append(' ') out.append(f' ') out.append(' ') out.append(f' {b64}') out.append(' ') out.append(' ') out.append(' ') elif btype == "caption": text = block["text"] out.append(f' {parse_inlines_to_odf(text)}') elif btype == "table": tbl_idx += 1 header = block["header"] rows = block["rows"] cols_count = len(header) out.append(f' ') out.append(f' ') out.append(' ') out.append(' ') for cell in header: out.append(' ') out.append(f' {parse_inlines_to_odf(cell)}') out.append(' ') out.append(' ') out.append(' ') for r_idx, row in enumerate(rows): cell_style = "TableCellDataAlt" if r_idx % 2 == 1 else "TableCellData" out.append(' ') for cell in row: out.append(f' ') out.append(f' {parse_inlines_to_odf(cell)}') out.append(' ') out.append(' ') out.append(' ') elif btype == "ordered_list": items = block["items"] out.append(' ') for item in items: out.append(' ') out.append(f' {parse_inlines_to_odf(item)}') out.append(' ') out.append(' ') elif btype == "unordered_list": items = block["items"] out.append(' ') for item in items: out.append(' ') out.append(f' {parse_inlines_to_odf(item)}') out.append(' ') out.append(' ') out.append(' ') out.append(' ') out.append('') fodt_content = "\n".join(out) with open(FODT_PATH, "w", encoding="utf-8") as f: f.write(fodt_content) print(f"FODT successfully written to {FODT_PATH} ({os.path.getsize(FODT_PATH)} bytes)") def generate_html_and_pdf(blocks): print("Generating Publication-Grade HTML and compiling to PDF via Playwright...") from playwright.sync_api import sync_playwright html = [] html.append('') html.append('') html.append('') html.append(' ') html.append(' PANDUAN SISTEM LENGKAP: RETRAINING, ANOTASI & LIVE COUNTING KARUNG KONVEYOR') html.append(' ') html.append('') html.append('') first_h1 = True in_meta_section = False for block in blocks: btype = block["type"] if btype == "hr": if in_meta_section: in_meta_section = False elif btype == "heading": level = block["level"] text = block["text"] if level == 1: if first_h1 and "PANDUAN SISTEM LENGKAP" in text: html.append('
') html.append('
') html.append(f'

{parse_inlines_to_html(text)}

') first_h1 = False else: html.append(f'

{parse_inlines_to_html(text)}

') elif level == 2: if text == "Daftar Isi": html.append('
') html.append('
Daftar Isi
') else: html.append(f'

{parse_inlines_to_html(text)}

') elif level == 3: if text == "Informasi Dokumen": in_meta_section = True html.append('
') else: html.append(f'

{parse_inlines_to_html(text)}

') else: html.append(f'

{parse_inlines_to_html(text)}

') elif btype == "paragraph": text = block["text"] if text.startswith("**Dokumentasi Arsitektur"): html.append(f'

{parse_inlines_to_html(text)}

') else: html.append(f'

{parse_inlines_to_html(text)}

') elif btype == "admonition": text = block["text"] cls = "admonition-warning" if ("Perhatian" in text or "Peringatan" in text) else "admonition-info" html.append(f'
{parse_inlines_to_html(text)}
') elif btype == "code": code_text = block["content"] html.append(f'
{escape_html(code_text)}
') elif btype == "image": src = block["src"] alt = block["alt"] img_info = get_image_info(src) data_uri = img_info["data_uri"] html.append('
') html.append(f' {escape_html(alt)}') html.append('
') elif btype == "caption": text = block["text"] html.append(f'
{parse_inlines_to_html(text)}
') elif btype == "table": header = block["header"] rows = block["rows"] html.append('') html.append(' ') for cell in header: html.append(f' ') html.append(' ') html.append(' ') for row in rows: html.append(' ') for cell in row: html.append(f' ') html.append(' ') html.append(' ') html.append('
{parse_inlines_to_html(cell)}
{parse_inlines_to_html(cell)}
') elif btype == "ordered_list": items = block["items"] # Check if this is the Table of Contents list # If so, close toc-card html.append('
    ' if "Bab 1:" in items[0] else '
      ') for item in items: html.append(f'
    1. {parse_inlines_to_html(item)}
    2. ') html.append('
    ') if "Bab 1:" in items[0]: html.append('
') # close toc-card html.append('
') # close cover-container elif btype == "unordered_list": items = block["items"] if in_meta_section: # Format as meta cards for item in items: parts = item.split(":", 1) if len(parts) == 2: lbl, val = parts html.append(f'
{parse_inlines_to_html(lbl.strip("* "))}: {parse_inlines_to_html(val.strip())}
') else: html.append(f'
{parse_inlines_to_html(item)}
') html.append('
') # close meta-grid in_meta_section = False else: html.append('') html.append('') html.append('') html_content = "\n".join(html) # Save debug html debug_html_path = os.path.join(WORKSPACE_ROOT, "docs/panduan_debug.html") with open(debug_html_path, "w", encoding="utf-8") as f: f.write(html_content) print(f"Debug HTML written to {debug_html_path}") # Compile with Playwright header_tpl = ( '
' 'PANDUAN SISTEM LENGKAP: RETRAINING, ANOTASI & LIVE COUNTING KARUNG KONVEYOR' '
' ) footer_tpl = ( '
' 'Dokumentasi Teknis Produksi v4.2.0' 'Halaman dari ' '
' ) with sync_playwright() as p: browser = p.chromium.launch() page = browser.new_page() page.set_content(html_content, wait_until="networkidle") page.pdf( path=PDF_PATH, format="A4", print_background=True, display_header_footer=True, header_template=header_tpl, footer_template=footer_tpl, margin={ "top": "22mm", "bottom": "22mm", "left": "18mm", "right": "18mm" } ) browser.close() print(f"PDF successfully compiled to {PDF_PATH} ({os.path.getsize(PDF_PATH)} bytes)") def main(): print("=== STARTING FULL PUBLICATION BUILD ===") if not os.path.exists(MD_PATH): raise FileNotFoundError(f"Source markdown not found: {MD_PATH}") with open(MD_PATH, "r", encoding="utf-8") as f: content = f.read() blocks = parse_markdown_blocks(content) print(f"Parsed {len(blocks)} markdown blocks.") # 1. Generate FODT generate_fodt(blocks) # 2. Generate PDF generate_html_and_pdf(blocks) print("=== BUILD COMPLETED SUCCESSFULLY ===") if __name__ == "__main__": main()