Files
reTraining/scripts/verify_panduan_md.py

176 lines
6.6 KiB
Python

#!/usr/bin/env python3
"""
Comprehensive Verification Suite for docs/PANDUAN_SISTEM_LENGKAP.md
Tests structure, 14 chapters, 21 figures, diagram assets, network ports,
zero AI clichés, zero em-dashes, and Indonesian technical lexicon.
"""
import os
import re
import sys
TARGET_FILE = "/home/asus/feedmill/reTraining/docs/PANDUAN_SISTEM_LENGKAP.md"
def test_verification():
assert os.path.exists(TARGET_FILE), f"File {TARGET_FILE} does not exist!"
with open(TARGET_FILE, "r", encoding="utf-8") as f:
content = f.read()
lines = content.splitlines()
print(f"=== VERIFICATION AUDIT FOR {TARGET_FILE} ===")
print(f"Total Lines: {len(lines)}")
print(f"Total Characters: {len(content)}")
print(f"Total Words: {len(content.split())}")
assert len(content) > 25000, f"File too short: {len(content)} characters"
# 1. Verify all 14 chapters
print("\n--- 1. Testing 14 Chapters Structure ---")
for i in range(1, 15):
pattern = rf"^# Bab {i}:"
matches = [line for line in lines if re.match(pattern, line)]
assert len(matches) == 1, f"Chapter {i} missing or duplicated: found {len(matches)}"
print(f" [PASS] Bab {i}: {matches[0]}")
# 2. Verify all 21 screenshot figures
print("\n--- 2. Testing 21 Screenshot References & Captions ---")
expected_screenshots = [
"01_projects_page.png",
"02_project_create_modal.png",
"03_library_video_archive.png",
"04_trim_page.png",
"05_batches_page.png",
"06_batches_sam3_auto_annotate_modal.png",
"07_batches_mass_auto_annotate_modal.png",
"08_review_annotation_canvas.png",
"09_review_filmstrip_quick_reclass.png",
"10_review_triage_crop_grid.png",
"11_review_triage_scatter_plot.png",
"12_review_exemplar_pool_panel.png",
"13_data_prep_quality_outliers.png",
"14_data_prep_augmentation_panel.png",
"15_data_prep_merge_target_modal.png",
"16_datasets_page.png",
"17_models_training_page.png",
"18_counting_bench_page.png",
"19_live_count_page.png",
"20_sam3_playground_page.png",
"21_workflow_progress_states.png",
]
for shot in expected_screenshots:
assert shot in content, f"Screenshot {shot} not referenced in document!"
print(f" [PASS] Screenshot referenced: {shot}")
for i in range(1, 22):
caption_pattern = rf"\*Gambar {i}:"
assert re.search(caption_pattern, content), f"Caption Gambar {i}: not found!"
print(f" [PASS] Caption verified: Gambar {i}")
# 3. Verify Diagram Deliverables
print("\n--- 3. Testing Architecture Diagram References ---")
assert "diagram-alur.png" in content, "diagram-alur.png reference missing!"
assert "diagram-alur.svg" in content, "diagram-alur.svg reference missing!"
assert "diagram-alur.fodg" in content, "diagram-alur.fodg reference missing!"
print(" [PASS] All diagram deliverables referenced (PNG, SVG, FODG).")
# 4. Verify Network Ports
print("\n--- 4. Testing Network Port Allocations ---")
ports = ["8080", "8000", "5173", "5000", "8554", "8889"]
for p in ports:
assert p in content, f"Port {p} missing from document!"
print(f" [PASS] Port {p} documented.")
# 5. Natural Language QC: Banned AI Clichés
print("\n--- 5. Testing Natural Language QC: Banned AI Clichés ---")
banned_phrases = [
r"mari kita jelajahi",
r"mari kita bahas",
r"penting untuk dicatat bahwa",
r"secara keseluruhan",
r"sebagai kesimpulan",
r"dalam lanskap teknologi",
r"dengan kata lain",
r"\btentunya\b",
r"harap diingat bahwa",
r"patut dicatat",
r"solusi mutakhir",
r"tidak diragukan lagi bahwa"
]
banned_violations = []
for bp in banned_phrases:
matches = re.findall(bp, content, re.IGNORECASE)
if matches:
banned_violations.append((bp, len(matches)))
if banned_violations:
for v, cnt in banned_violations:
print(f" [FAIL] Found banned phrase '{v}' ({cnt} occurrences)")
sys.exit(1)
else:
print(" [PASS] Zero banned AI clichés detected (100% clean).")
# 6. Natural Language QC: Zero Em-Dashes (--, —) outside code blocks, thematic breaks & table dividers
print("\n--- 6. Testing Natural Language QC: Zero Em-Dashes ---")
prose_lines = []
in_code_block = False
for line in lines:
if line.strip().startswith("```"):
in_code_block = not in_code_block
continue
if in_code_block:
continue
if line.strip() == "---":
continue
# ignore markdown table separator rows like |---|---|
if "|" in line and "-" in line and not any(c.isalnum() for c in line.replace("|", "").replace("-", "").replace(":", "")):
continue
prose_lines.append(line)
dash_violations = []
for line_idx, line in enumerate(prose_lines, 1):
if "—" in line:
dash_violations.append((line_idx, "—", line))
if "--" in line:
# check if inside inline code like `--host`
line_no_code = re.sub(r"`[^`]*`", "", line)
if "--" in line_no_code:
dash_violations.append((line_idx, "--", line))
if dash_violations:
for idx, char, line in dash_violations[:10]:
print(f" [FAIL] Dash violation on prose line {idx}: found '{char}' in '{line.strip()}'")
sys.exit(1)
else:
print(" [PASS] Zero em-dashes (-- or —) detected in prose (100% clean).")
# 7. Verify Standardized Indonesian Technical Lexicon
print("\n--- 7. Testing Standardized Technical Lexicon ---")
lexicon = [
"inferensi",
"anotasi",
"pelabelan otomatis",
"bobot",
"penyetelan halus",
"data acuan kebenaran",
"tulang punggung",
"kepala penyelaras",
"pembagian validasi stabil",
"filter pencilan",
"augmentasi",
"snapshot",
"serah-terima lintasan",
"deteksi semu",
"penghitungan berlebih",
"penghitungan kurang",
"kunci eksklusif GPU"
]
for term in lexicon:
assert term.lower() in content.lower(), f"Technical term '{term}' not found in document!"
print(f" [PASS] Term verified: '{term}'")
print("\n=======================================================")
print("ALL VERIFICATION CHECKS PASSED (100% SPEC COMPLIANCE)!")
print("=======================================================")
if __name__ == "__main__":
test_verification()