# ==========================================
# OCR CELLS
# ==========================================

import os
import json
import cv2

from paddleocr import PaddleOCR

ROWS_DIR = r"G:\00. PROJECTS\API\Inventory\output\rows"

# ==========================================
# INIT OCR
# ==========================================

print("Loading PaddleOCR...")

ocr = PaddleOCR(
    use_angle_cls=True,
    lang="en"
)

print("Ready")
print("")

# ==========================================
# DETAIL COLUMNS
# ==========================================

DETAIL_COLUMNS = [

    "NamaBarang",
    "Jumlah",
    "Satuan",
    "HargaBeli",
    "Diskon",
    "PPn",
    "SubTotal",
    "Barcode"

]

# ==========================================
# GET ROW COUNT
# ==========================================

row_count = 0

for file_name in os.listdir(ROWS_DIR):

    if file_name.startswith("NamaBarang_"):

        row_count += 1

print("ROW COUNT =", row_count)
print("")

# ==========================================
# OCR SINGLE CELL
# ==========================================

def read_cell(file_name):

    if not os.path.exists(file_name):

        return ""

    try:

        img = cv2.imread(file_name)

        if img is None:

            return ""

        # ==================================
        # UPSCALE
        # ==================================

        img = cv2.resize(
            img,
            None,
            fx=4,
            fy=4,
            interpolation=cv2.INTER_CUBIC
        )

        # ==================================
        # GRAYSCALE
        # ==================================

        gray = cv2.cvtColor(
            img,
            cv2.COLOR_BGR2GRAY
        )

        # ==================================
        # OTSU
        # ==================================

        _, thresh = cv2.threshold(
            gray,
            0,
            255,
            cv2.THRESH_BINARY + cv2.THRESH_OTSU
        )

        # ==================================
        # OCR
        # ==================================

        result = ocr.ocr(
            thresh,
            cls=True
        )

        if result is None:

            return ""

        if len(result) == 0:

            return ""

        if result[0] is None:

            return ""

        texts = []

        for line in result[0]:

            try:

                texts.append(
                    line[1][0]
                )

            except:

                pass

        return " ".join(texts)

    except Exception as ex:

        print(
            f"ERROR OCR : {file_name}"
        )

        print(ex)

        return ""

# ==========================================
# BUILD JSON
# ==========================================

rows = []

for row_no in range(
    1,
    row_count + 1
):

    print(
        f"ROW {row_no:03d}"
    )

    item = {}

    for col_name in DETAIL_COLUMNS:

        file_name = os.path.join(
            ROWS_DIR,
            f"{col_name}_{row_no:03d}.jpg"
        )

        text = read_cell(
            file_name
        )

        item[col_name] = text

        print(
            f"   {col_name:12} = {text}"
        )

    rows.append(
        item
    )

    print("")

# ==========================================
# RESULT
# ==========================================

print("")
print("===== DETAIL JSON =====")
print("")

print(
    json.dumps(
        rows,
        indent=2,
        ensure_ascii=False
    )
)

print("")
print("======================")
print("")
print("SELESAI")