"""PDF Extractor configuration — columns, extraction prompt, Groq client factory."""
import os

OUTPUT_COLS = [
    "Invoice No", "Invoice Date", "PO#", "Line#",
    "Model / Item#", "Description", "Qty", "Unit Price", "Amount", "COO",
]

OUTPUT_FIELDS = [
    "invoice_no", "invoice_date", "po_no", "line_no",
    "model_item_no", "description", "qty", "unit_price", "amount", "coo",
]

FIELD_LABELS = OUTPUT_COLS

MODELS = [
    "llama-3.3-70b-versatile",
    "llama3-8b-8192",
]

EXTRACT_PROMPT = """
Extract all line items from this commercial invoice. Return JSON only.

For each line item find:
- invoice_no: main invoice/document number
- invoice_date: invoice date
- po_no: Customer PO number — labeled as "Customer PO#" or "Customer PO Number". Repeat per row.
- line_no: line/item/pos/seq number
- model_item_no: part number/material number/SKU — short code only, never a description sentence
- description: product name only, no codes or part numbers
- qty: quantity number only
- unit_price: price per unit
- amount: total line value (Extension/Extended Value/Ext.Price/Net Amount). Never use unit_price as amount.
- coo: country of origin — 2-letter code or country name only

Rules:
- Extract every line item; skip metadata rows (RPM, HTS, ECCN, weight rows)
- Repeat invoice_no, invoice_date on every row
- If a field is missing use ""

Return ONLY:
{
  "vendor_name": "",
  "line_items": [
    {
      "invoice_no": "", "invoice_date": "", "po_no": "",
      "line_no": "", "model_item_no": "", "description": "",
      "qty": "", "unit_price": "", "amount": "", "coo": ""
    }
  ]
}

Invoice text:
{text}
"""


def get_groq_client():
    api_key = os.getenv("GROQ_API_KEY", "")
    if not api_key:
        return None
    try:
        from groq import Groq
        return Groq(api_key=api_key)
    except ImportError:
        return None


# Singleton
groq_client = get_groq_client()