Files
pfm-ocr/backend/sources/generate_side_by_side.py
T
fhanyuh caf8e98378 chore: normalize line endings (CRLF -> LF)
No content changes: git diff --ignore-all-space over these files is empty.
The churn came from editing on Windows against a repo checked out with LF.
2026-08-27 10:40:49 +07:00

161 lines
6.5 KiB
Python

import json
import os
def clean_val(val):
if val is None:
return ""
s = str(val).strip().upper()
if s in ["N/A", "NOT FOUND", "NOTFOUND", "EMPTY", "NONE", "-", "N / A"]:
return ""
s = " ".join(s.split())
s = s.replace("PT. ", "PT.")
s = s.replace("✓", "").replace("✔", "").strip()
return s
def main():
ai_file = "backend/sources/ai_results.json"
manual_file = "backend/sources/manual_labels.json"
output_md = "backend/sources/comparison_side_by_side.md"
if not os.path.exists(ai_file) or not os.path.exists(manual_file):
print("Required files not found.")
return
with open(ai_file, "r", encoding="utf-8") as f:
ai_data = json.load(f)
with open(manual_file, "r", encoding="utf-8") as f:
manual_data = json.load(f)
ai_dict = {item.get("filename"): item for item in ai_data if item.get("filename")}
manual_dict = {item.get("filename"): item for item in manual_data if item.get("filename")}
common_filenames = sorted(list(set(ai_dict.keys()) & set(manual_dict.keys())))
header_fields = [
("noPO", "noPO", "PO Number"),
("noSO", "noSO", "SO Number"),
("noDO", "noDO", "DO Number"),
("tanggal", "tanggal", "Date"),
("plat", "platTruk", "Plat Nomor"),
("customer", "customerInfo", "Customer Name"),
("store", "orderUntuk", "Store Name"),
("alamat", "alamat", "Alamat")
]
stats = {
"PO Number": {"match": 0, "total": 0},
"SO Number": {"match": 0, "total": 0},
"DO Number": {"match": 0, "total": 0},
"Date": {"match": 0, "total": 0},
"Plat Nomor": {"match": 0, "total": 0},
"Customer Name": {"match": 0, "total": 0},
"Store Name": {"match": 0, "total": 0},
"Alamat": {"match": 0, "total": 0},
"Item SKU": {"match": 0, "total": 0},
"Item Banyak": {"match": 0, "total": 0},
"Item Jumlah": {"match": 0, "total": 0}
}
md_content = []
md_content.append("# Side-by-Side Accuracy Comparison Report\n\n")
md_content.append("<!-- summary_placeholder -->\n\n")
md_content.append("## Detailed Comparison per File\n\n")
for filename in common_filenames:
manual = manual_dict[filename]
ai = ai_dict[filename]
ai_meta = ai.get("layer3Final", {})
md_content.append(f"### 📄 `{filename}`\n\n")
# Header fields table
md_content.append("#### Header Fields\n")
md_content.append("| Field | AI Value (OCR) | Manual Value (Ground Truth) | Status |\n")
md_content.append("|---|---|---|---|\n")
for manual_key, ai_key, field_label in header_fields:
m_val = clean_val(manual.get(manual_key))
a_val = clean_val(ai_meta.get(ai_key))
is_match = (m_val == a_val)
status = "✅ Match" if is_match else "❌ Mismatch"
stats[field_label]["total"] += 1
if is_match:
stats[field_label]["match"] += 1
md_content.append(f"| {field_label} | `{a_val or '(empty)'}` | `{m_val or '(empty)'}` | {status} |\n")
md_content.append("\n")
# Items comparison
m_items = manual.get("items", [])
ai_items = ai.get("items", [])
if not ai_items and "items" in ai_meta:
ai_items = ai_meta.get("items", [])
m_items_dict = {clean_val(item.get("kodeBarang")): item for item in m_items if clean_val(item.get("kodeBarang"))}
ai_items_dict = {clean_val(item.get("kodeBarang")): item for item in ai_items if clean_val(item.get("kodeBarang"))}
all_skus = sorted(list(set(m_items_dict.keys()) | set(ai_items_dict.keys())))
md_content.append("#### Items List\n")
md_content.append("| SKU | AI Banyak | Manual Banyak | Banyak Status | AI Jumlah | Manual Jumlah | Jumlah Status |\n")
md_content.append("|---|---|---|---|---|---|---|\n")
for sku in all_skus:
m_item = m_items_dict.get(sku)
ai_item = ai_items_dict.get(sku)
sku_match = (m_item is not None) and (ai_item is not None)
stats["Item SKU"]["total"] += 1
if sku_match:
stats["Item SKU"]["match"] += 1
m_banyak = clean_val(m_item.get("banyak")) if m_item else ""
ai_banyak = clean_val(ai_item.get("banyak")) if ai_item else ""
banyak_match = (m_banyak == ai_banyak)
banyak_status = "✅ Match" if banyak_match else "❌ Mismatch"
stats["Item Banyak"]["total"] += 1
if banyak_match:
stats["Item Banyak"]["match"] += 1
m_jumlah = clean_val(m_item.get("jumlah")) if m_item else ""
ai_jumlah = clean_val(ai_item.get("jumlah")) if ai_item else ""
jumlah_match = (m_jumlah == ai_jumlah)
jumlah_status = "✅ Match" if jumlah_match else "❌ Mismatch"
stats["Item Jumlah"]["total"] += 1
if jumlah_match:
stats["Item Jumlah"]["match"] += 1
md_content.append(f"| `{sku}` | `{ai_banyak or '(empty)'}` | `{m_banyak or '(empty)'}` | {banyak_status} | `{ai_jumlah or '(empty)'}` | `{m_jumlah or '(empty)'}` | {jumlah_status} |\n")
md_content.append("\n---\n\n")
# Generate summary block
sum_md = []
sum_md.append("### Summary Accuracy Metrics\n\n")
sum_md.append("| Field / Area | Total Checks | Matches | Mismatches | Accuracy (%) |\n")
sum_md.append("|---|---|---|---|---|\n")
total_matches = 0
total_fields = 0
for field_label, counts in stats.items():
match_cnt = counts["match"]
total_cnt = counts["total"]
pct = (match_cnt / total_cnt * 100.0) if total_cnt > 0 else 100.0
sum_md.append(f"| {field_label} | {total_cnt} | {match_cnt} | {total_cnt - match_cnt} | {pct:.2f}% |\n")
total_matches += match_cnt
total_fields += total_cnt
overall_accuracy = (total_matches / total_fields * 100.0) if total_fields > 0 else 100.0
sum_md.append(f"| **OVERALL TOTAL** | **{total_fields}** | **{total_matches}** | **{total_fields - total_matches}** | **{overall_accuracy:.2f}%** |\n")
full_md = "".join(md_content).replace("<!-- summary_placeholder -->", "".join(sum_md))
with open(output_md, "w", encoding="utf-8") as f:
f.write(full_md)
print(f"Side-by-side report generated at: {output_md}")
if __name__ == "__main__":
main()