# ==========================================
# PARSER HEADER
# ==========================================

import re


# ==========================================
# NORMALIZE DATE
# ==========================================

def normalize_date(text):

    if not text:
        return ""

    text = text.strip()

    # OCR typo umum
    text = (
        text
        .replace("Stp", "Sep")
        .replace("5tp", "Sep")
        .replace("5ep", "Sep")
        .replace("0ct", "Oct")
        .replace("/", "-")
    )

    return text


# ==========================================
# NORMALIZE NUMBER
# ==========================================

def normalize_number(text):

    if not text:
        return ""

    text = text.strip()

    # hapus spasi
    text = text.replace(" ", "")

    # OCR kadang menghasilkan:
    # 750.000
    # 750,000
    # 750.000,00

    text = text.replace(".", "")
    text = text.replace(",", "")

    if text == "":
        return ""

    return text


# ==========================================
# NORMALIZE TEXT
# ==========================================

def normalize_text(
    field_name,
    text
):

    if text is None:
        return ""

    text = text.strip()

    # ==========================
    # FIELD TANGGAL
    # ==========================

    if field_name in (
        "TglTx",
        "Overdue"
    ):

        return normalize_date(text)

    # ==========================
    # FIELD ANGKA
    # ==========================

    if field_name in (
        "Total",
        "Diskon",
        "Net",
        "PPn",
        "SubTotal",
        "DiskonLain",
        "BiayaLain",
        "GrandTotal",
        "TPPn",
        "TDiskon"
    ):

        return normalize_number(text)

    return text