#!/usr/bin/env python3
"""
Publication-Grade Document Compiler for LibreOffice Flat XML (.fodt) and PDF.
Milestone 3 - Retraining & Live Counting Documentation
"""
import os
import sys
import re
import base64
import xml.etree.ElementTree as ET
from PIL import Image
WORKSPACE_ROOT = "/home/asus/feedmill/reTraining"
MD_PATH = os.path.join(WORKSPACE_ROOT, "docs/PANDUAN_SISTEM_LENGKAP.md")
FODT_PATH = os.path.join(WORKSPACE_ROOT, "docs/PANDUAN_SISTEM_LENGKAP.fodt")
PDF_PATH = os.path.join(WORKSPACE_ROOT, "docs/PANDUAN_SISTEM_LENGKAP.pdf")
def escape_xml(s):
if s is None:
return ""
return (s.replace("&", "&")
.replace("<", "<")
.replace(">", ">")
.replace('"', """)
.replace("'", "'"))
def escape_html(s):
if s is None:
return ""
return (s.replace("&", "&")
.replace("<", "<")
.replace(">", ">")
.replace('"', """))
def get_image_info(rel_path):
# Determine full path
if rel_path.startswith("docs/"):
full_path = os.path.join(WORKSPACE_ROOT, rel_path)
elif os.path.exists(os.path.join(WORKSPACE_ROOT, "docs", rel_path)):
full_path = os.path.join(WORKSPACE_ROOT, "docs", rel_path)
elif os.path.exists(os.path.join(WORKSPACE_ROOT, rel_path)):
full_path = os.path.join(WORKSPACE_ROOT, rel_path)
else:
raise FileNotFoundError(f"Image not found: {rel_path}")
with open(full_path, "rb") as f:
data = f.read()
b64_str = base64.b64encode(data).decode("ascii")
with Image.open(full_path) as img:
w_px, h_px = img.size
# Printable width in A4 is 17.0 cm (21.0 - 2.0 - 2.0)
aspect = w_px / h_px
if aspect > 1.7:
# Diagram (16:9)
width_cm = 16.5
height_cm = width_cm / aspect
else:
# 16:10 screenshots
width_cm = 15.8
height_cm = width_cm / aspect
return {
"full_path": full_path,
"base64": b64_str,
"width_px": w_px,
"height_px": h_px,
"width_cm": round(width_cm, 2),
"height_cm": round(height_cm, 2),
"data_uri": f"data:image/png;base64,{b64_str}"
}
def parse_inlines_to_odf(text):
# Tokenize inline formatting: code, bolditalic, bold, italic, links, math
pattern = re.compile(r"(`[^`]+`|\*\*\*[^*]+\*\*\*|\*\*[^*]+\*\*|\*[^*]+\*|\[[^\]]+\]\([^)]+\)|\$[^$]+\$)")
pos = 0
result = []
for m in pattern.finditer(text):
if m.start() > pos:
result.append(escape_xml(text[pos:m.start()]))
token = m.group(0)
if token.startswith("`") and token.endswith("`"):
code_text = escape_xml(token[1:-1])
result.append(f'\1', text)
# Bold italic
text = re.sub(r"\*\*\*([^*]+)\*\*\*", r"\1", text)
# Bold
text = re.sub(r"\*\*([^*]+)\*\*", r"\1", text)
# Italic
text = re.sub(r"\*([^*]+)\*", r"\1", text)
# Links
text = re.sub(r"\[([^\]]+)\]\(([^)]+)\)", r'\1', text)
# Math var
text = re.sub(r"\$([^$]+)\$", r'\1', text)
return text
def parse_markdown_blocks(content):
lines = content.split("\n")
blocks = []
i = 0
n = len(lines)
while i < n:
line = lines[i]
stripped = line.strip()
# Empty line
if not stripped:
i += 1
continue
# Horizontal Rule
if stripped in ["---", "***", "___"]:
blocks.append({"type": "hr"})
i += 1
continue
# Code block
if stripped.startswith("```"):
lang = stripped[3:].strip()
code_lines = []
i += 1
while i < n and not lines[i].strip().startswith("```"):
code_lines.append(lines[i])
i += 1
if i < n: # consume closing ```
i += 1
blocks.append({"type": "code", "lang": lang, "content": "\n".join(code_lines)})
continue
# Headings
if stripped.startswith("#"):
level = 0
while level < len(stripped) and stripped[level] == "#":
level += 1
title = stripped[level:].strip()
blocks.append({"type": "heading", "level": level, "text": title})
i += 1
continue
# Table
if stripped.startswith("|") and stripped.endswith("|"):
table_lines = []
while i < n and lines[i].strip().startswith("|") and lines[i].strip().endswith("|"):
table_lines.append(lines[i].strip())
i += 1
if len(table_lines) >= 2:
header = [c.strip() for c in table_lines[0].split("|")[1:-1]]
rows = []
for r in table_lines[2:]:
cols = [c.strip() for c in r.split("|")[1:-1]]
rows.append(cols)
blocks.append({"type": "table", "header": header, "rows": rows})
continue
# Image
img_match = re.match(r"^!\[(.*?)\]\((.*?)\)$", stripped)
if img_match:
alt, src = img_match.groups()
blocks.append({"type": "image", "alt": alt, "src": src})
i += 1
continue
# Numbered list: 1.
num_match = re.match(r"^(\d+)\.\s+(.*)$", stripped)
if num_match:
list_items = []
while i < n:
curr_stripped = lines[i].strip()
nm = re.match(r"^(\d+)\.\s+(.*)$", curr_stripped)
if nm:
list_items.append(nm.group(2))
i += 1
else:
break
blocks.append({"type": "ordered_list", "items": list_items})
continue
# Bullet list: - or *
bullet_match = re.match(r"^[-*]\s+(.*)$", stripped)
if bullet_match and not re.match(r"^\*[^*]+\*$", stripped):
bullet_items = []
while i < n:
curr_stripped = lines[i].strip()
bm = re.match(r"^[-*]\s+(.*)$", curr_stripped)
if bm and not re.match(r"^\*[^*]+\*$", curr_stripped):
bullet_items.append(bm.group(1))
i += 1
else:
break
blocks.append({"type": "unordered_list", "items": bullet_items})
continue
# Callout / Admonition (*Perhatian:...* or *Invarian penting:...*)
if stripped.startswith("*Perhatian:") or stripped.startswith("*Invarian penting:") or stripped.startswith("*Catatan:") or stripped.startswith("*Peringatan:"):
blocks.append({"type": "admonition", "text": stripped})
i += 1
continue
# Caption or italic paragraph under image (*Gambar X:...* or *Unduh format...*)
if stripped.startswith("*Gambar ") or stripped.startswith("*Unduh format vektor"):
blocks.append({"type": "caption", "text": stripped})
i += 1
continue
# Standard paragraph
p_lines = [stripped]
i += 1
while i < n:
next_line = lines[i]
next_stripped = next_line.strip()
if (not next_stripped or
next_stripped.startswith("#") or
next_stripped.startswith("```") or
next_stripped.startswith("|") or
next_stripped.startswith("---") or
next_stripped.startswith("![") or
re.match(r"^\d+\.\s+", next_stripped) or
re.match(r"^[-*]\s+", next_stripped) or
next_stripped.startswith("*Perhatian:") or
next_stripped.startswith("*Invarian penting:") or
next_stripped.startswith("*Gambar ") or
next_stripped.startswith("*Unduh format")):
break
p_lines.append(next_stripped)
i += 1
blocks.append({"type": "paragraph", "text": " ".join(p_lines)})
return blocks
print("Parser module loaded successfully.")
def generate_fodt(blocks):
print("Generating OASIS OpenDocument Text Flat XML (.fodt)...")
# Header
out = []
out.append('')
out.append('