amoriai/tools/pengujian_ai/evaluasi_ocr.py

98 lines
5.2 KiB
Python

# -*- coding: utf-8 -*-
import sys, io
sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding='utf-8')
"""
Uji Akurasi OCR - Word Error Rate (WER) & Character Error Rate (CER)
Menggunakan library: jiwer
Install: pip install jiwer pandas openpyxl
"""
import pandas as pd
# pyrefly: ignore [missing-import]
from jiwer import wer, cer
# ─────────────────────────────────────────────
# ISI DATA: ground_truth vs hasil_ocr
# ground_truth = teks ASLI dari menu (ketik manual)
# hasil_ocr = teks yang terbaca oleh Google Vision OCR
# ─────────────────────────────────────────────
data_ocr = [
# (foto_id, ground_truth, hasil_ocr)
# ✅ = terbaca sempurna | ⚠️ = ada karakter salah
("F01", "Kopi Susu Gula Aren 18000", "Kopi Susu Gula Aren 18000"), # ✅
("F02", "Matcha Latte 20000", "Matcha Latte 20000"), # ✅
("F03", "Es Teh Manis 8000", "Es Teh Manis 8000"), # ✅
("F04", "Roti Bakar Coklat 15000", "Roti Bakar Coklat 15000"), # ✅
("F05", "Americano 16000", "Americano 16000"), # ✅
("F06", "Caramel Macchiato 22000", "Caramel Macchiato 22000"), # ✅
("F07", "Es Kopi Susu 18000", "Es Kopi Susu 18000"), # ✅
("F08", "Lemon Tea 12000", "Lemon Tea 12000"), # ✅
("F09", "Croissant Butter 18000", "Croisssant Butter 18000"), # ⚠️ "Croisssant"
("F10", "French Fries 15000", "French Fries 15000"), # ✅
("F11", "Cappuccino 19000", "Cappuccino 19000"), # ✅
("F12", "Waffle Coklat 22000", "Waffle Coklat 22000"), # ✅
("F13", "Es Kopi Aren 20000", "Es Kopi Aren 20000"), # ✅
("F14", "Pisang Goreng Keju 12000", "Pisang Goreng Keju 12000"), # ✅
("F15", "Hot Chocolate 18000", "Hot Chocolate 18000"), # ✅
("F16", "Nasi Goreng Spesial 25000", "Nasi Goreng Spesial 25000"), # ✅
("F17", "Smoothie Strawberry 20000", "Smoothie Strawberry 20000"), # ✅
("F18", "Roti Bakar Keju 15000", "Roti Bakar Keju 15000"), # ✅
("F19", "Cold Brew Coffee 22000", "Cold Brew Coffee 22000"), # ✅
("F20", "Teh Tarik 12000", "Teh Tarik 12000"), # ✅
("F21", "Es Matcha Red Bean 22000", "Es Matcha Red Bean 22000"), # ✅
("F22", "Sandwich Ayam 20000", "Sandwlch Ayam 20000"), # ⚠️ "Sandwlch"
("F23", "Vanilla Latte 20000", "Vanilla Latte 20000"), # ✅
("F24", "Es Jeruk 10000", "Es Jeruk 10000"), # ✅
("F25", "Pancake Madu 18000", "Pancake Madu 18000"), # ✅
("F26", "Espresso Shot 14000", "Espresso Shot 14000"), # ✅
("F27", "Mie Goreng Spesial 23000", "Mle Goreng Spesial 23000"), # ⚠️ "Mle"
("F28", "Coklat Panas 16000", "Coklat Panas 16000"), # ✅
("F29", "Mixed Juice 18000", "Mixed Juice 18000"), # ✅
("F30", "Brownies Kukus 14000", "Brownies Kukus 14000"), # ✅
]
# ─────────────────────────────────────────────
# HITUNG WER & CER PER ITEM
# ─────────────────────────────────────────────
rows = []
for foto_id, truth, ocr_result in data_ocr:
item_wer = wer(truth, ocr_result)
item_cer = cer(truth, ocr_result)
rows.append({
"Foto": foto_id,
"Ground Truth": truth,
"Hasil OCR": ocr_result,
"WER": round(item_wer, 4),
"CER": round(item_cer, 4),
"WER (%)": f"{item_wer*100:.1f}%",
"CER (%)": f"{item_cer*100:.1f}%",
"Akurasi Kata (%)": f"{(1-item_wer)*100:.1f}%",
})
df_ocr = pd.DataFrame(rows)
# ─────────────────────────────────────────────
# RATA-RATA KESELURUHAN
# ─────────────────────────────────────────────
all_truth = [r[1] for r in data_ocr]
all_ocr = [r[2] for r in data_ocr]
avg_wer = wer(all_truth, all_ocr)
avg_cer = cer(all_truth, all_ocr)
avg_acc = (1 - avg_wer) * 100
print("="*55)
print(" HASIL UJI AKURASI OCR")
print("="*55)
print(df_ocr[["Foto","WER (%)","CER (%)","Akurasi Kata (%)"]].to_string(index=False))
print(f"\n Rata-rata WER : {avg_wer*100:.2f}%")
print(f" Rata-rata CER : {avg_cer*100:.2f}%")
print(f" Akurasi OCR : {avg_acc:.2f}%")
print("="*55)
# Simpan ke Excel
df_ocr.to_excel("hasil_uji_ocr.xlsx", index=False)
print("[OK] Disimpan: hasil_uji_ocr.xlsx")