import pdfplumber

INVOICE_KEYWORDS = [
    'invoice', 'commercial invoice',
    'bill to', 'ship to', 'invoice date', 'purchase order',
    'unit price', 'ext. price', 'item no', 'material no',
    'amount', 'quantity'
]

SKIP_PAGE_KEYWORDS = [
    'terms of sale', 'terms and conditions', 'limited warranty',
    'indemnification', 'dispute resolution', 'miscellaneous',
    'air waybill', 'packing list', 'certificate of conformance',
    'shipper certifies', 'conditions of contract',
    'delivery order',
]

LINE_ITEM_KEYWORDS = ['unit price', 'unit value', 'extended price', 'extended value', 'ext. price', 'qty uom']

def is_invoice_page(text: str, page_num: int = 0) -> bool:
    text_lower = text.lower()
    if any(kw in text_lower for kw in SKIP_PAGE_KEYWORDS):
        if page_num == 0:
            return True
        if any(kw in text_lower for kw in LINE_ITEM_KEYWORDS):
            return True
        return False
    if any(kw in text_lower for kw in LINE_ITEM_KEYWORDS):
        return True
    return any(kw in text_lower for kw in INVOICE_KEYWORDS)

def extract_text_from_pdf(pdf_file) -> str:
    text = ""
    try:
        with pdfplumber.open(pdf_file) as pdf:
            for i, page in enumerate(pdf.pages[:15]):
                t = page.extract_text()
                if not t:
                    continue
                if is_invoice_page(t, page_num=i):
                    text += f"\n--- PAGE {i+1} ---\n"
                    text += t + "\n"
                    print(f"Page {i+1}: INCLUDED ({len(t)} chars)")
                else:
                    print(f"Page {i+1}: SKIPPED")
    except Exception as e:
        raise Exception(f"PDF read error: {str(e)}")

    print(f"TOTAL TEXT: {len(text)} chars")
    print(f"Page {i+1}: INCLUDED ({len(t)} chars) — first 50: {t[:50]}")
    return text