import re
from parsers.utils import extract_raw_text


def parse(pdf_path, vendor):

    raw_text = extract_raw_text(pdf_path)

    invoice_number = ""
    purchase_order = ""
    total_amount = ""

    # =========================
    # 1️⃣ Invoice Number
    # Under "Invoice Number:"
    # =========================
    match = re.search(
        r"Invoice\s*Number:\s*\n?\s*(\d+)",
        raw_text,
        re.IGNORECASE
    )
    if match:
        invoice_number = match.group(1)

    # =========================
    # 2️⃣ Purchase Order
    # Capture PO: PO000135710
    # =========================
    match = re.search(
        r"PO:\s*(PO\d+)",
        raw_text,
        re.IGNORECASE
    )
    if match:
        purchase_order = match.group(1)
    else:
        # Fallback: from Order / Purchase Order table
        match = re.search(
            r"Purchase\s+Order\s+(PO\d+)",
            raw_text,
            re.IGNORECASE
        )
        if match:
            purchase_order = match.group(1)

    # =========================
    # 3️⃣ Total Amount
    # Match USD Total
    # =========================
    match = re.search(
        r"USD\s*Total\s*([\d,]+\.\d{2})",
        raw_text,
        re.IGNORECASE
    )
    if match:
        total_amount = match.group(1)

    return {
        "vendor_name": vendor,
        "invoice_number": invoice_number,
        "purchase_order": purchase_order,
        "total_amount_usd": f"${total_amount}" if total_amount else "",
        "status": "Processed" if invoice_number else "Pending"
    }