452 lines
19 KiB
Python
452 lines
19 KiB
Python
import json
|
|
import os
|
|
import pandas as pd
|
|
from openpyxl.styles import Font, PatternFill, Alignment, Border, Side
|
|
from openpyxl.utils import get_column_letter
|
|
|
|
def clean_val(val):
|
|
if val is None:
|
|
return ""
|
|
s = str(val).strip().upper()
|
|
if s in ["N/A", "NOT FOUND", "NOTFOUND", "EMPTY", "NONE", "-", "N / A"]:
|
|
return ""
|
|
s = " ".join(s.split())
|
|
s = s.replace("PT. ", "PT.")
|
|
s = s.replace("✓", "").replace("✔", "").strip()
|
|
return s
|
|
|
|
def main():
|
|
ai_file = "sources/ai_results.json"
|
|
manual_file = "sources/manual_labels.json"
|
|
xlsx_file = "sources/comparison_report.xlsx"
|
|
images_dir = "sources/test-images"
|
|
|
|
if not os.path.exists(ai_file):
|
|
# Fallback to backend/sources
|
|
ai_file = "backend/sources/ai_results.json"
|
|
manual_file = "backend/sources/manual_labels.json"
|
|
xlsx_file = "backend/sources/comparison_report.xlsx"
|
|
images_dir = "backend/sources/test-images"
|
|
|
|
if not os.path.exists(ai_file):
|
|
print(f"Error: AI results file not found at {ai_file}")
|
|
return
|
|
|
|
if not os.path.exists(manual_file):
|
|
print(f"Error: Manual labels file not found at {manual_file}")
|
|
return
|
|
|
|
# Load data
|
|
with open(ai_file, "r", encoding="utf-8") as f:
|
|
ai_data = json.load(f)
|
|
|
|
with open(manual_file, "r", encoding="utf-8") as f:
|
|
manual_data = json.load(f)
|
|
|
|
# Convert to dict for lookup by filename
|
|
ai_dict = {item.get("filename"): item for item in ai_data if item.get("filename")}
|
|
manual_dict = {item.get("filename"): item for item in manual_data if item.get("filename")}
|
|
|
|
print(f"Loaded {len(ai_dict)} AI results from file.")
|
|
print(f"Loaded {len(manual_dict)} manual labels from file.")
|
|
|
|
# Scan for physical image files in test-images folder
|
|
existing_images = None
|
|
if os.path.exists(images_dir):
|
|
existing_images = set(os.listdir(images_dir))
|
|
print(f"Found {len(existing_images)} physical images in '{images_dir}'.")
|
|
else:
|
|
print(f"Warning: Images directory not found at '{images_dir}'.")
|
|
|
|
# Find mismatches in file lists
|
|
only_in_ai = set(ai_dict.keys()) - set(manual_dict.keys())
|
|
only_in_manual = set(manual_dict.keys()) - set(ai_dict.keys())
|
|
if only_in_ai:
|
|
print(f"Warning: {len(only_in_ai)} files exist only in AI results: {only_in_ai}")
|
|
if only_in_manual:
|
|
print(f"Warning: {len(only_in_manual)} files exist only in Manual labels: {only_in_manual}")
|
|
|
|
# Determine files to compare (must exist in AI results, Manual labels, and physically as images if directory is available)
|
|
common_filenames = set(ai_dict.keys()) & set(manual_dict.keys())
|
|
|
|
if existing_images is not None:
|
|
deleted_images = common_filenames - existing_images
|
|
if deleted_images:
|
|
print(f"Info: Excluded {len(deleted_images)} files that were physically deleted from images folder: {deleted_images}")
|
|
all_filenames = sorted(list(common_filenames & existing_images))
|
|
else:
|
|
all_filenames = sorted(list(common_filenames))
|
|
|
|
print(f"Comparing {len(all_filenames)} matching images.")
|
|
|
|
header_fields = [
|
|
("noPO", "noPO", "PO Number"),
|
|
("noSO", "noSO", "SO Number"),
|
|
("noDO", "noDO", "DO Number"),
|
|
("tanggal", "tanggal", "Date"),
|
|
("plat", "platTruk", "Plat Nomor"),
|
|
("customer", "customerInfo", "Customer Name"),
|
|
("store", "orderUntuk", "Store Name"),
|
|
("alamat", "alamat", "Alamat")
|
|
]
|
|
|
|
doc_comparison_rows = []
|
|
item_comparison_rows = []
|
|
|
|
# Counters for accuracy calculation
|
|
stats = {
|
|
"PO Number": {"match": 0, "total": 0},
|
|
"SO Number": {"match": 0, "total": 0},
|
|
"DO Number": {"match": 0, "total": 0},
|
|
"Date": {"match": 0, "total": 0},
|
|
"Plat Nomor": {"match": 0, "total": 0},
|
|
"Customer Name": {"match": 0, "total": 0},
|
|
"Store Name": {"match": 0, "total": 0},
|
|
"Alamat": {"match": 0, "total": 0},
|
|
"Item SKU": {"match": 0, "total": 0},
|
|
"Item Banyak": {"match": 0, "total": 0},
|
|
"Item Jumlah": {"match": 0, "total": 0}
|
|
}
|
|
|
|
# Document-level side-by-side rows
|
|
doc_side_by_side_rows = []
|
|
|
|
for filename in all_filenames:
|
|
manual = manual_dict.get(filename)
|
|
ai = ai_dict.get(filename)
|
|
|
|
if not manual:
|
|
print(f"Warning: Manual label not found for {filename} (exists only in AI results)")
|
|
continue
|
|
if not ai:
|
|
print(f"Warning: AI result not found for {filename} (exists only in Manual labels)")
|
|
continue
|
|
|
|
ai_meta = ai.get("layer3Final", {})
|
|
|
|
# 1. Compare header fields (Vertical format for filtering)
|
|
sxs_row = {"Filename": filename}
|
|
for manual_key, ai_key, field_label in header_fields:
|
|
m_val = clean_val(manual.get(manual_key))
|
|
a_val = clean_val(ai_meta.get(ai_key))
|
|
is_match = (m_val == a_val)
|
|
|
|
doc_comparison_rows.append({
|
|
"Filename": filename,
|
|
"Field": field_label,
|
|
"AI Value (OCR)": a_val if a_val else "(empty)",
|
|
"Manual Value (Ground Truth)": m_val if m_val else "(empty)",
|
|
"Match": "Match" if is_match else "Mismatch"
|
|
})
|
|
|
|
# Side-by-side
|
|
sxs_row[f"{field_label} (AI)"] = a_val if a_val else ""
|
|
sxs_row[f"{field_label} (Manual)"] = m_val if m_val else ""
|
|
sxs_row[f"{field_label} Status"] = "Match" if is_match else "Mismatch"
|
|
|
|
stats[field_label]["total"] += 1
|
|
if is_match:
|
|
stats[field_label]["match"] += 1
|
|
|
|
doc_side_by_side_rows.append(sxs_row)
|
|
|
|
# 2. Compare items
|
|
m_items = manual.get("items", [])
|
|
ai_items = ai.get("items", [])
|
|
if not ai_items and "items" in ai_meta:
|
|
ai_items = ai_meta.get("items", [])
|
|
|
|
# Create dictionaries of items indexed by codeBarang (SKU)
|
|
m_items_dict = {clean_val(item.get("kodeBarang")): item for item in m_items if clean_val(item.get("kodeBarang"))}
|
|
ai_items_dict = {clean_val(item.get("kodeBarang")): item for item in ai_items if clean_val(item.get("kodeBarang"))}
|
|
|
|
# Check all unique SKUs across both manual and AI
|
|
all_skus = set(list(m_items_dict.keys()) + list(ai_items_dict.keys()))
|
|
|
|
for sku in all_skus:
|
|
m_item = m_items_dict.get(sku)
|
|
ai_item = ai_items_dict.get(sku)
|
|
|
|
# Check SKU existence match
|
|
sku_match = (m_item is not None) and (ai_item is not None)
|
|
stats["Item SKU"]["total"] += 1
|
|
if sku_match:
|
|
stats["Item SKU"]["match"] += 1
|
|
|
|
m_banyak = clean_val(m_item.get("banyak")) if m_item else ""
|
|
ai_banyak = clean_val(ai_item.get("banyak")) if ai_item else ""
|
|
banyak_match = (m_banyak == ai_banyak)
|
|
|
|
stats["Item Banyak"]["total"] += 1
|
|
if banyak_match:
|
|
stats["Item Banyak"]["match"] += 1
|
|
|
|
m_jumlah = clean_val(m_item.get("jumlah")) if m_item else ""
|
|
ai_jumlah = clean_val(ai_item.get("jumlah")) if ai_item else ""
|
|
jumlah_match = (m_jumlah == ai_jumlah)
|
|
|
|
stats["Item Jumlah"]["total"] += 1
|
|
if jumlah_match:
|
|
stats["Item Jumlah"]["match"] += 1
|
|
|
|
# Log code comparison
|
|
item_comparison_rows.append({
|
|
"Filename": filename,
|
|
"Kode Barang (SKU)": sku,
|
|
"Field": "SKU Existence",
|
|
"AI Value (OCR)": sku if ai_item else "(not found)",
|
|
"Manual Value (Ground Truth)": sku if m_item else "(not found)",
|
|
"Match": "Match" if sku_match else "Mismatch"
|
|
})
|
|
|
|
# Log Banyak comparison
|
|
item_comparison_rows.append({
|
|
"Filename": filename,
|
|
"Kode Barang (SKU)": sku,
|
|
"Field": "Banyak (Qty Package)",
|
|
"AI Value (OCR)": ai_banyak if ai_banyak else "(empty)",
|
|
"Manual Value (Ground Truth)": m_banyak if m_banyak else "(empty)",
|
|
"Match": "Match" if banyak_match else "Mismatch"
|
|
})
|
|
|
|
# Log Jumlah comparison
|
|
item_comparison_rows.append({
|
|
"Filename": filename,
|
|
"Kode Barang (SKU)": sku,
|
|
"Field": "Jumlah (Qty Unit)",
|
|
"AI Value (OCR)": ai_jumlah if ai_jumlah else "(empty)",
|
|
"Manual Value (Ground Truth)": m_jumlah if m_jumlah else "(empty)",
|
|
"Match": "Match" if jumlah_match else "Mismatch"
|
|
})
|
|
|
|
# Prepare summary data
|
|
summary_rows = []
|
|
total_matches = 0
|
|
total_fields = 0
|
|
for field_label, counts in stats.items():
|
|
match_cnt = counts["match"]
|
|
total_cnt = counts["total"]
|
|
pct = (match_cnt / total_cnt * 100.0) if total_cnt > 0 else 100.0
|
|
summary_rows.append({
|
|
"Field / Area": field_label,
|
|
"Total Checks": total_cnt,
|
|
"Matches": match_cnt,
|
|
"Mismatches": total_cnt - match_cnt,
|
|
"Accuracy (%)": round(pct, 2)
|
|
})
|
|
total_matches += match_cnt
|
|
total_fields += total_cnt
|
|
|
|
overall_accuracy = (total_matches / total_fields * 100.0) if total_fields > 0 else 100.0
|
|
summary_rows.append({
|
|
"Field / Area": "OVERALL TOTAL",
|
|
"Total Checks": total_fields,
|
|
"Matches": total_matches,
|
|
"Mismatches": total_fields - total_matches,
|
|
"Accuracy (%)": round(overall_accuracy, 2)
|
|
})
|
|
|
|
df_summary = pd.DataFrame(summary_rows)
|
|
df_docs = pd.DataFrame(doc_comparison_rows)
|
|
df_sxs = pd.DataFrame(doc_side_by_side_rows)
|
|
df_items = pd.DataFrame(item_comparison_rows)
|
|
|
|
# Styling setup
|
|
font_family = "Segoe UI"
|
|
header_font = Font(name=font_family, size=11, bold=True, color="FFFFFF")
|
|
regular_font = Font(name=font_family, size=10)
|
|
bold_font = Font(name=font_family, size=10, bold=True)
|
|
title_font = Font(name=font_family, size=16, bold=True, color="1F4E78")
|
|
|
|
header_fill = PatternFill(start_color="1F4E78", end_color="1F4E78", fill_type="solid") # Dark Blue
|
|
zebra_fill = PatternFill(start_color="F2F5F8", end_color="F2F5F8", fill_type="solid") # Zebra light blue-gray
|
|
match_fill = PatternFill(start_color="E2EFDA", end_color="E2EFDA", fill_type="solid") # Light green
|
|
mismatch_fill = PatternFill(start_color="FCE4D6", end_color="FCE4D6", fill_type="solid") # Light orange
|
|
|
|
center_align = Alignment(horizontal="center", vertical="center")
|
|
left_align = Alignment(horizontal="left", vertical="center")
|
|
right_align = Alignment(horizontal="right", vertical="center")
|
|
|
|
thin_side = Side(border_style="thin", color="D9D9D9")
|
|
cell_border = Border(left=thin_side, right=thin_side, top=thin_side, bottom=thin_side)
|
|
|
|
# Save to Excel
|
|
os.makedirs(os.path.dirname(xlsx_file), exist_ok=True)
|
|
|
|
writer = None
|
|
for attempt in range(1, 10):
|
|
try:
|
|
writer = pd.ExcelWriter(xlsx_file, engine='openpyxl')
|
|
break
|
|
except PermissionError:
|
|
base_dir = os.path.dirname(xlsx_file)
|
|
filename = os.path.basename(xlsx_file)
|
|
name, ext = os.path.splitext(filename)
|
|
if "_" in name and name.split("_")[-1].isdigit():
|
|
name = "_".join(name.split("_")[:-1])
|
|
xlsx_file = os.path.join(base_dir, f"{name}_{attempt}{ext}")
|
|
|
|
if writer is None:
|
|
print("Error: Could not open the Excel writer because the file is locked.")
|
|
return
|
|
|
|
with writer:
|
|
df_summary.to_excel(writer, sheet_name='Summary Accuracy', index=False, startrow=3)
|
|
df_sxs.to_excel(writer, sheet_name='Side-by-Side Comparison', index=False)
|
|
df_docs.to_excel(writer, sheet_name='Header Field Comparison', index=False)
|
|
df_items.to_excel(writer, sheet_name='Item SKU Comparison', index=False)
|
|
|
|
# 1. Style Summary Sheet with a Title Banner
|
|
ws_sum = writer.sheets['Summary Accuracy']
|
|
ws_sum.views.sheetView[0].showGridLines = True
|
|
ws_sum.cell(row=1, column=1, value="AI OCR vs. Manual Ground Truth Accuracy Report").font = title_font
|
|
ws_sum.row_dimensions[1].height = 30
|
|
|
|
# Style Summary Table Headers
|
|
max_col_sum = df_summary.shape[1]
|
|
for col in range(1, max_col_sum + 1):
|
|
cell = ws_sum.cell(row=4, column=col)
|
|
cell.font = header_font
|
|
cell.fill = header_fill
|
|
cell.alignment = center_align
|
|
cell.border = cell_border
|
|
|
|
# Style Summary Data
|
|
max_row_sum = ws_sum.max_row
|
|
for row in range(5, max_row_sum + 1):
|
|
for col in range(1, max_col_sum + 1):
|
|
cell = ws_sum.cell(row=row, column=col)
|
|
cell.font = regular_font
|
|
cell.border = cell_border
|
|
if col == 1:
|
|
cell.alignment = left_align
|
|
else:
|
|
cell.alignment = right_align
|
|
|
|
# Zebra style
|
|
if row % 2 == 0 and row != max_row_sum:
|
|
cell.fill = zebra_fill
|
|
|
|
# Format percentage
|
|
if col == 5 and isinstance(cell.value, (int, float)):
|
|
cell.number_format = '0.00"%"'
|
|
|
|
# Bold overall total row
|
|
if row == max_row_sum:
|
|
for col in range(1, max_col_sum + 1):
|
|
c = ws_sum.cell(row=row, column=col)
|
|
c.font = bold_font
|
|
c.fill = match_fill if overall_accuracy > 80 else mismatch_fill
|
|
|
|
# Auto-adjust column width for Summary
|
|
for col in ws_sum.columns:
|
|
max_len = max(len(str(cell.value or '')) for cell in col)
|
|
col_letter = get_column_letter(col[0].column)
|
|
ws_sum.column_dimensions[col_letter].width = max(max_len + 4, 12)
|
|
|
|
# Style detail worksheets
|
|
for sheet_name in ['Side-by-Side Comparison', 'Header Field Comparison', 'Item SKU Comparison']:
|
|
ws = writer.sheets[sheet_name]
|
|
ws.views.sheetView[0].showGridLines = True
|
|
max_row = ws.max_row
|
|
max_col = ws.max_column
|
|
|
|
# Header row styling
|
|
for col in range(1, max_col + 1):
|
|
cell = ws.cell(row=1, column=col)
|
|
cell.font = header_font
|
|
cell.fill = header_fill
|
|
cell.alignment = center_align
|
|
cell.border = cell_border
|
|
|
|
# Data rows styling
|
|
for row in range(2, max_row + 1):
|
|
is_zebra = (row % 2 == 0)
|
|
|
|
# For Side-by-Side Comparison
|
|
if sheet_name == 'Side-by-Side Comparison':
|
|
for col in range(1, max_col + 1):
|
|
cell = ws.cell(row=row, column=col)
|
|
cell.font = regular_font
|
|
cell.border = cell_border
|
|
|
|
if col == 1:
|
|
cell.alignment = left_align
|
|
if is_zebra:
|
|
cell.fill = zebra_fill
|
|
else:
|
|
# Apply alignments and color mismatch/match
|
|
# Format of headers:
|
|
# Col 1: Filename
|
|
# Col 2: PO AI, Col 3: PO Manual, Col 4: PO Status
|
|
# ... and so on
|
|
# So status is at index col where (col - 1) % 3 == 0 (4, 7, 10, 13, 16, 19, 22, 25)
|
|
col_pos = col - 1
|
|
if col_pos % 3 == 0: # This is a Status column
|
|
status_val = cell.value
|
|
cell.alignment = center_align
|
|
if status_val == "Match":
|
|
cell.fill = match_fill
|
|
else:
|
|
cell.fill = mismatch_fill
|
|
else: # This is AI or Manual value column
|
|
cell.alignment = left_align
|
|
# Match background of the cell with its corresponding status cell (two columns to the right if AI, one if Manual)
|
|
status_col_idx = col + (2 if col_pos % 3 == 1 else 1)
|
|
status_val = ws.cell(row=row, column=status_col_idx).value
|
|
if status_val == "Match":
|
|
if is_zebra:
|
|
# Let's keep it subtle
|
|
pass
|
|
else:
|
|
# Highlight mismatches clearly
|
|
cell.fill = mismatch_fill
|
|
|
|
# For vertical comparison sheets
|
|
else:
|
|
# Match column is the last column
|
|
match_cell = ws.cell(row=row, column=max_col)
|
|
match_val = match_cell.value
|
|
|
|
for col in range(1, max_col + 1):
|
|
cell = ws.cell(row=row, column=col)
|
|
cell.font = regular_font
|
|
cell.border = cell_border
|
|
|
|
# Apply alignments based on column
|
|
if col in [1, 2, 3, 4]:
|
|
cell.alignment = left_align
|
|
else:
|
|
cell.alignment = center_align
|
|
|
|
# Color match / mismatch
|
|
if match_val == "Match":
|
|
cell.fill = match_fill
|
|
elif match_val == "Mismatch":
|
|
cell.fill = mismatch_fill
|
|
elif is_zebra:
|
|
cell.fill = zebra_fill
|
|
|
|
# Auto-fit columns
|
|
for col in ws.columns:
|
|
max_len = 0
|
|
for cell in col:
|
|
val_str = str(cell.value or '')
|
|
# Limit long text like Alamat from making column excessively wide
|
|
if sheet_name == 'Side-by-Side Comparison' and cell.column in [22, 23]: # Alamat
|
|
max_len = max(max_len, min(len(val_str), 30))
|
|
elif sheet_name == 'Header Field Comparison' and cell.column in [3, 4]: # Values
|
|
max_len = max(max_len, min(len(val_str), 40))
|
|
else:
|
|
max_len = max(max_len, len(val_str))
|
|
col_letter = get_column_letter(col[0].column)
|
|
ws.column_dimensions[col_letter].width = max(max_len + 4, 12)
|
|
|
|
print("\n=== Accuracy Report Summary ===")
|
|
print(df_summary.to_string(index=False))
|
|
print("===============================\n")
|
|
print(f"Comparison report generated at {xlsx_file}")
|
|
|
|
if __name__ == "__main__":
|
|
main()
|