"""
Clickdealer in stock reader.

Reads Steven's own in stock list from myclickdealer.co.uk (stock_list.php,
status ins) and counts vehicles per make and model, so the cockpit can flag an
auction car as filling a gap or as a model he already holds plenty of.

The list is a classic server rendered table. Each data row carries the VRM and
a combined "MAKE MODEL" cell (for example VAUXHALL CORSA, BMW 1 SERIES,
MERCEDES-BENZ A-CLASS). Recon captured the real layout 2026-06-10, see
data/inspect/clickdealer_tab0.html.

Model matching: Clickdealer, Motorway and Carwow name the same car differently
(MERCEDES-BENZ A-CLASS versus Mercedes A 180). model_key() normalises both
sides through CompMatch's own canonical grouping (bid_suggest.model_key), so
MERCEDES-BENZ A-CLASS and Mercedes A Class both key to "mercedes a", and a BMW
118D keys to "bmw 1" alongside a BMW 1 Series. This is an annotation, never a
gate, so an occasional mismatch only mislabels a badge.

This reads a report screen only. It never changes anything on Clickdealer.
"""

import datetime
import re

from .. import bid_suggest, browser

# A VRM in the row identifies a real vehicle row rather than a header or totals
# line. Current style AB12CDE plus older prefix styles.
_VRM_RE = re.compile(r"[A-Z]{2}\d{2}\s?[A-Z]{3}|[A-Z]\d{1,3}\s?[A-Z]{3}")

_ROW_RE = re.compile(r"<tr[^>]*>(.*?)</tr>", re.S)
_CELL_RE = re.compile(r"<td[^>]*>(.*?)</td>", re.S)
_TAG_RE = re.compile(r"<[^>]+>")


def _norm(text):
    """Lower case, hyphens to spaces, collapsed whitespace."""
    return " ".join(str(text or "").lower().replace("-", " ").split())


def model_key(make, model):
    """The shared make+model key used on BOTH sides of the stock match, and
    by the sold history note and the top picks. CompMatch's own canonical
    grouping (bid_suggest.model_key) lower cased, reused rather than copied
    so all three land on exactly the same group.

    Examples: ("MERCEDES-BENZ", "A-CLASS") -> "mercedes a",
    ("Vauxhall", "Corsa Energy AC") -> "vauxhall corsa",
    ("BMW", "1 Series Sport") and ("BMW", "118D M Sport") -> "bmw 1".

    Before 2026-09-20 this took the first token of the make plus the first
    token of the model on its own, which split one real model in two
    wherever a platform prints the engine badge as the model word: a BMW
    118D keyed to "bmw 118d" and never met the "bmw 1" stock count, a
    Mercedes A 180 never met an A-CLASS, an "MG Motor UK ZS" keyed to "mg
    motor", and Motorway's truncated "Insig" never met an Insignia. So a
    model that was actually sitting in stock read as a gap, and the top
    picks, which score partly on that gap, inherited it. CompMatch had
    already solved the same problem for the suggested bid.

    An empty make or an empty model still keys to "" (never a guess)."""
    if not _norm(make) or not _norm(model):
        return ""
    # Hyphens inside the MODEL read as spaces, as they always did here, so a
    # Honda CR-V and a Honda CR V still meet. The make keeps its own hyphen,
    # which is how bid_suggest.model_key recognises MERCEDES-BENZ.
    return bid_suggest.model_key(f"{make} {_norm(model)}").lower()


def _cells(row_html):
    return [" ".join(_TAG_RE.sub(" ", t).split()).strip() for t in _CELL_RE.findall(row_html)]


def parse_stock(html):
    """Parse the stock_list.php report into {model_key: count} plus the raw
    (make, model) pairs for the snapshot. Returns (counts, pairs). A vehicle row
    is one with a VRM in its early cells and the combined make/model cell two
    cells after the VRM (Keytag, ?, Stock#, VRM, Registered Date, Make/Model)."""
    counts = {}
    pairs = []
    for row in _ROW_RE.findall(html):
        c = _cells(row)
        if len(c) < 7:
            continue
        vrm_idx = None
        for i, cell in enumerate(c[:6]):
            if cell and _VRM_RE.fullmatch(cell):
                vrm_idx = i
                break
        if vrm_idx is None:
            continue
        mm = c[vrm_idx + 2] if vrm_idx + 2 < len(c) else ""
        toks = mm.split()
        if len(toks) < 2:
            continue
        make, model = toks[0], " ".join(toks[1:])
        key = model_key(make, model)
        if not key:
            continue
        counts[key] = counts.get(key, 0) + 1
        pairs.append((make, model))
    return counts, pairs


def _money(text):
    """'12,399.00' -> 12399 (whole pounds). None when not a number."""
    t = str(text or "").replace(",", "").replace("&pound;", "").replace("£", "").strip()
    try:
        return int(round(float(t)))
    except ValueError:
        return None


def _date_iso(text):
    """'05/05/2026' -> '2026-05-05'. None when not a date."""
    m = re.fullmatch(r"(\d{2})/(\d{2})/(\d{4})", str(text or "").strip())
    return f"{m.group(3)}-{m.group(2)}-{m.group(1)}" if m else None


def _int(text):
    try:
        return int(str(text or "").replace(",", "").strip())
    except ValueError:
        return None


def parse_sold(html):
    """Parse the sold vehicles report. Each vehicle row is anchored on its VRM
    cell; the columns sit at fixed offsets from it (verified against the live
    May 2026 report). Returns a list of dicts, one per sold car, including the
    vehicle_details id so the supplier can be read from the car's own page."""
    out = []
    for row in _ROW_RE.findall(html):
        c = _cells(row)
        if len(c) < 25:
            continue
        vrm_idx = None
        for i, cell in enumerate(c[:8]):
            if cell and _VRM_RE.fullmatch(cell):
                vrm_idx = i
                break
        if vrm_idx is None or vrm_idx < 3:
            continue
        vid = re.search(r"/vehicle_details\.php\?id=(\d+)", row)
        try:
            out.append({
                "stock_number": c[vrm_idx - 3],
                "reg": c[vrm_idx],
                "make": c[vrm_idx + 3],
                "model": c[vrm_idx + 4],
                "fuel": c[vrm_idx + 8],
                "mileage": _int(c[vrm_idx + 9]),
                "sale_price": _money(c[vrm_idx + 10]),
                "purchase_price": _money(c[vrm_idx + 11]),
                # TotalMargin = car margin plus additional profit (finance etc.),
                # before prep is deducted. Steven's headline margin is this
                # minus the SIV, computed at ingest.
                "total_margin": _money(c[vrm_idx + 14]),
                "sold_date": _date_iso(c[vrm_idx + 18]),
                "days_to_sell": _int(c[vrm_idx + 21]),
                "vehicle_id": vid.group(1) if vid else None,
            })
        except IndexError:
            continue
    return [r for r in out if r["sold_date"]]


def parse_margins(html):
    """Parse the sales margins report into {stock_number: {prep_total, margin}}.
    Anchored on the VRM cell: SIV (prep total) is six cells after it, Margin
    twelve after (verified live: net sales minus purchase minus SIV equals the
    Margin column)."""
    out = {}
    for row in _ROW_RE.findall(html):
        c = _cells(row)
        if len(c) < 22:
            continue
        vrm_idx = None
        for i, cell in enumerate(c[:12]):
            if cell and _VRM_RE.fullmatch(cell):
                vrm_idx = i
                break
        if vrm_idx is None or vrm_idx < 1:
            continue
        stock_no = c[0]
        if not stock_no:
            continue
        try:
            out[stock_no] = {
                "prep_total": _money(c[vrm_idx + 6]),
                "margin": _money(c[vrm_idx + 12]),
            }
        except IndexError:
            continue
    return out


# Expense types that are buying and moving costs, not prep work. Steven wants
# the reported prep figure to be actual prep only (repairs, bodywork, tyres,
# MOT and so on). These still come out of the margin, they just are not "prep".
# VAT reclaim and repayment lines are accounting adjustments, also not prep.
NON_PREP_TYPES = ("buyers fee", "shipping", "delivery", "collection", "vat ")


def is_prep_type(expense_type):
    t = (expense_type or "").strip().lower()
    return bool(t) and not any(t.startswith(x) for x in NON_PREP_TYPES)


def actual_prep(split):
    """The actual prep spend from a per car expense split, fees excluded.
    None when there is no split to read."""
    if not split:
        return None
    return sum(v for t, v in split.items() if is_prep_type(t))


def parse_siv(html):
    """Parse a car's own SIV page (siv.php?id=N, the screen behind the SIV
    button on the vehicle page) into a prep split by expense type plus the
    total. Expense rows read: date, supplier, type, description, amount, View.
    Returns (split_dict, total) where split is {type: pounds}."""
    split = {}
    total = None
    for row in _ROW_RE.findall(html):
        c = _cells(row)
        if len(c) >= 4 and c[2] == "Total SIV":
            total = _money(c[3])
            continue
        if len(c) >= 5 and re.search(r"\d+\.\d\d", c[4] or ""):
            typ = (c[2] or "").strip()
            amount = _money(c[4])
            if typ and amount is not None:
                split[typ] = split.get(typ, 0) + amount
    return split, total


def parse_supplier(html):
    """The supplier a car was bought from, read from the selected option of the
    purchase form's supplier dropdown on vehicle_details.php."""
    m = re.search(r'<select[^>]*name="v\[supplier\]"[^>]*>(.*?)</select>', html, re.S)
    if not m:
        return None
    sel = re.search(r"<option[^>]*selected[^>]*>([^<]*)</option>", m.group(1))
    return sel.group(1).strip() or None if sel else None


def _month_range(year, month):
    import calendar
    last = calendar.monthrange(year, month)[1]
    return last


def read_sales_month(playwright, year, month, headless=True, progress=None):
    """Read one month of sold cars: the sold vehicles report joined to the
    sales margins report by stock number, plus each car's supplier from its own
    page. Returns a list of dicts ready for db.record_sales_history. Fails
    loudly if the reports do not render."""
    base = browser.SITES["clickdealer"]["base_url"]
    last = _month_range(year, month)
    ctx = browser.open_reader_context(playwright, "clickdealer", headless=headless)
    try:
        page = ctx.pages[0] if ctx.pages else ctx.new_page()

        sold_url = (f"{base}/sold_vehicles.php?start_day=1&start_month={month}"
                    f"&start_year={year}&end_day={last}&end_month={month}&end_year={year}"
                    f"&user_id=&sale_type=&location_id=&view=&action=View+Report")
        page.goto(sold_url, wait_until="domcontentloaded", timeout=60000)
        sold_html = page.content()
        if "SoldDate" not in sold_html and "Days" not in sold_html:
            raise RuntimeError(f"Sold vehicles report did not render for {year}-{month:02d}.")
        sold = parse_sold(sold_html)

        margins_url = (f"{base}/sales_margins.php?start_date=01/{month:02d}/{year}"
                       f"&end_date={last}/{month:02d}/{year}&complete_month=&user_id="
                       f"&sale_type=&view=&location_id=&supplier=&warranty_type_id="
                       f"&buyer_id=&action=View+Report")
        page.goto(margins_url, wait_until="domcontentloaded", timeout=60000)
        margins = parse_margins(page.content())

        for i, r in enumerate(sold, 1):
            m = margins.get(r["stock_number"], {})
            r["prep_total"] = m.get("prep_total")
            r["source"] = None
            r["prep_split"] = None
            if r.get("vehicle_id"):
                # The car's own pages: supplier from the purchase form, and the
                # prep split by type from its SIV screen (the page behind the
                # SIV button, confirmed by clicking it live).
                try:
                    page.goto(f"{base}/vehicle_details.php?id={r['vehicle_id']}",
                              wait_until="domcontentloaded", timeout=40000)
                    r["source"] = parse_supplier(page.content())
                except Exception:
                    pass
                try:
                    page.goto(f"{base}/siv.php?id={r['vehicle_id']}",
                              wait_until="domcontentloaded", timeout=40000)
                    split, siv_total = parse_siv(page.content())
                    if split:
                        r["prep_split"] = split
                    # Prefer the SIV page total when the margins report had none.
                    if r["prep_total"] is None:
                        r["prep_total"] = siv_total
                except Exception:
                    pass
            # Steven's headline margin: everything earned including finance
            # (TotalMargin), with prep paid for.
            if r.get("total_margin") is not None and r.get("prep_total") is not None:
                r["margin"] = r["total_margin"] - r["prep_total"]
            else:
                r["margin"] = None
            if progress:
                progress(i, len(sold))
        return sold
    finally:
        ctx.close()


def read_in_stock(playwright, headless=True):
    """Open Steven's Clickdealer in stock list and return {model_key: count}.
    Fails loudly if the report does not render, never guesses."""
    today = datetime.date.today()
    url = (f"{browser.SITES['clickdealer']['stock_url']}"
           f"&date_day={today.day}&date_month={today.month}&date_year={today.year}")
    ctx = browser.open_reader_context(playwright, "clickdealer", headless=headless)
    try:
        page = ctx.pages[0] if ctx.pages else ctx.new_page()
        page.goto(url, wait_until="domcontentloaded", timeout=60000)
        html = page.content()
        if "Make / Model" not in html:
            raise RuntimeError(
                "Clickdealer stock list did not render the expected report "
                f"(no Make / Model header at {page.url}). The page may have "
                "changed or the login may have expired. Run: python3 login.py clickdealer manual"
            )
        counts, pairs = parse_stock(html)
        if not counts:
            raise RuntimeError(
                "Clickdealer stock list rendered but no vehicle rows were read. "
                "The layout may have changed, not guessing."
            )
        return counts, pairs
    finally:
        ctx.close()


# ---------------------------------------------------------------------------
# A sold car's own page (2026-09-14, Steven: "go into clickdealer and search
# each car that you have on record as being sold and find the sales info of
# that car to fully back populate it, each car will have photos too, i want
# just one photo of each sold car in there, the first one"). vehicle_details
# .php?id=N is a server rendered form: every fact about the car sits in a
# named input, select or textarea (the supplier dropdown parse_supplier
# reads is one of them), and the photos are plain img tags. No captured
# copy of the page exists here, so the readers below take every field and
# every picture as they are, and the probe run reports what it found for
# checking before the full run.

_TAG_ATTR_RE = re.compile(r'([A-Za-z_:\-\[\]\.]+)\s*=\s*(?:"([^"]*)"|\'([^\']*)\'|([^\s"\'>]+))')
_INPUT_RE = re.compile(r"<input\b[^>]*>", re.I)
_SELECT_RE = re.compile(r"<select\b([^>]*)>(.*?)</select>", re.S | re.I)
_TEXTAREA_RE = re.compile(r"<textarea\b([^>]*)>(.*?)</textarea>", re.S | re.I)
_OPTION_RE = re.compile(r"<option\b([^>]*)>(.*?)</option>", re.S | re.I)
_IMG_RE = re.compile(r"<img\b[^>]*>", re.I)
_A_IMG_RE = re.compile(r"<a\b[^>]*href=[\"']([^\"']+\.(?:jpe?g|png|webp)(?:\?[^\"']*)?)[\"']", re.I)
# A car photo is a jpeg, png or webp; a gif on this site is always furniture
# (the probe of 2026-09-14 picked images/cal.gif, a calendar icon).
_IMAGE_EXT_RE = re.compile(r"\.(jpe?g|png|webp)(\?|$)", re.I)
# Pictures that are the site's own furniture, never the car.
_NOT_A_CAR = ("logo", "icon", "button", "btn", "spinner", "loading", "blank", "pixel",
              "spacer", "flag", "arrow", "avatar", "badge", "captcha", "banner", "placeholder", "no_image", "noimage", "no-image")


def _tag_attrs(tag):
    out = {}
    for m in _TAG_ATTR_RE.finditer(tag):
        out[m.group(1).lower()] = m.group(2) if m.group(2) is not None else (m.group(3) if m.group(3) is not None else m.group(4))
    return out


def _unescape(text):
    import html as _html
    return _html.unescape(_TAG_RE.sub("", text or "")).strip()


def parse_vehicle_fields(html):
    """Every named form field on a car's page and its value, {name: value}:
    an input's value (a checkbox or radio only when checked, never a button),
    a select's chosen option text, a textarea's text. Names are as the page
    has them (v[make], v[mileage] ...). Later fields with the same name win
    only when the earlier one was empty."""
    out = {}

    def put(name, value):
        if not name:
            return
        value = (value or "").strip()
        if name not in out or (not out[name] and value):
            out[name] = value

    for tag in _INPUT_RE.findall(html):
        a = _tag_attrs(tag)
        kind = (a.get("type") or "text").lower()
        if kind in ("submit", "button", "image", "reset", "file", "password"):
            continue
        if kind in ("checkbox", "radio") and "checked" not in tag.lower():
            continue
        put(a.get("name"), _unescape(a.get("value", "")))
    for attrs, body in _SELECT_RE.findall(html):
        name = _tag_attrs(attrs).get("name")
        chosen = ""
        for oattrs, otext in _OPTION_RE.findall(body):
            if re.search(r"\bselected\b", oattrs, re.I):
                chosen = _unescape(otext)
                break
        put(name, chosen)
    for attrs, body in _TEXTAREA_RE.findall(html):
        put(_tag_attrs(attrs).get("name"), _unescape(body))
    return out


def parse_vehicle_photos(html, page_url=""):
    """The car's pictures in page order as absolute urls: every img (src,
    data-src, data-original) and every link straight to an image file,
    less the site's own furniture (logos, icons, buttons). The first is
    the one Dealer OS shows."""
    from urllib.parse import urljoin
    seen, out = set(), []

    def add(src):
        if not src or src.startswith("data:"):
            return
        url = urljoin(page_url, src.strip())
        low = url.lower()
        if re.search(r"\.gif(\?|$)", low):
            return
        if not _IMAGE_EXT_RE.search(low) and not any(w in low for w in ("image", "photo", "picture")):
            return
        if any(w in low for w in _NOT_A_CAR):
            return
        if url not in seen:
            seen.add(url)
            out.append(url)

    for tag in _IMG_RE.findall(html):
        a = _tag_attrs(tag)
        add(a.get("data-src") or a.get("data-original") or a.get("src"))
    for href in _A_IMG_RE.findall(html):
        add(href)
    # The car's own pictures live on the images host under /vehicles/
    # (seen on the probe: images.clickdealer.co.uk/vehicles/<id>/...); those
    # come first, the page's other pictures after, so the first is the car.
    def car_first(url):
        low = url.lower()
        return 0 if ("/vehicles/" in low or "//images." in low) else 1
    return sorted(out, key=car_first)


def _field(fields, *words, text=False):
    """The first field whose name holds one of the words (in the order the
    words are given) and has a value, as (name, value); (None, None) if none.
    text: a value that is only a number is skipped (v[variant_id]=54668 is
    an id, not the derivative; the probe of 2026-09-14 took it)."""
    low = {k: k.lower() for k in fields}
    for w in words:
        for name, name_low in low.items():
            val = fields[name]
            if w in name_low and val and not (text and re.fullmatch(r"[\d.,\s]+", val)):
                return name, val.strip()
    return None, None


def _number(text):
    m = re.search(r"\d[\d,]*(?:\.\d+)?", text or "")
    if not m:
        return None
    try:
        return float(m.group(0).replace(",", ""))
    except ValueError:
        return None


def vehicle_details(fields, photos):
    """What the Sold tab shows, read off a car's page fields: {year,
    derivative, litres, fuel, gearbox, mileage, colour, body, photo, keys}.
    keys names the field each fact came from, for checking. Anything not
    found is None."""
    keys = {}
    out = {}

    def take(fact, words, conv=None, text=False):
        name, val = _field(fields, *words, text=text)
        if name is None:
            out[fact] = None
            return
        v = conv(val) if conv else val
        out[fact] = v
        if v is not None and v != "":
            keys[fact] = name

    def year(val):
        m = re.search(r"\b(19[89]\d|20[0-4]\d)\b", val or "")
        return int(m.group(1)) if m else None

    def litres(val):
        n = _number(val)
        if n is None:
            return None
        if n > 50:
            return f"{round(n / 100) / 10:.1f}"
        return f"{n:.1f}"

    def miles(val):
        n = _number(val)
        return int(n) if n is not None and n >= 0 else None

    take("year", ("year_of_manufacture", "manufacture_year", "model_year", "reg_year", "year"), year)
    if out["year"] is None:
        take("year", ("date_of_registration", "registration_date", "reg_date", "first_registered", "registered"), year)
    take("derivative", ("edition", "derivative", "variant", "trim", "description", "title"), text=True)
    take("litres", ("engine_size", "engine_cc", "engine", "capacity", "cc"), litres)
    plain = lambda v: v.title() if v.isupper() else v
    take("fuel", ("fuel",), plain, text=True)
    take("gearbox", ("transmission", "gearbox"), plain, text=True)
    take("mileage", ("mileage", "odometer", "miles"), miles)
    take("colour", ("colour", "color"), lambda v: v.title() if v.isupper() else v, text=True)
    take("body", ("body_type", "bodytype", "body_style"), text=True)
    out["photo"] = photos[0] if photos else None
    out["keys"] = keys
    return out


def sold_car_report(base, details, probe_keys=None):
    """base (clickdealer_sale_report's plain row) filled in from the car's
    page: the name becomes "2018 Ford Kuga 1.5 TDCi Titanium 5dr", the spec
    "1.5 · Diesel · Manual", plus mileage, colour and the first photo (a
    url, the caller swaps it for Dealer OS's lasting link). record_keys
    names where each fact came from; on a probe it carries every field and
    picture found so the reader can be checked."""
    rep = dict(base)
    d = details
    bits = [str(d["year"]) if d.get("year") else None, rep.get("make"), rep.get("model"), d.get("derivative")]
    rep["name"] = " ".join(x for x in bits if x) or rep.get("name")
    spec = " · ".join(x for x in (d.get("litres"), d.get("fuel"), d.get("gearbox")) if x)
    rep["spec"] = spec or None
    rep["mileage"] = d.get("mileage")
    rep["colour"] = d.get("colour") or None
    rep["photo"] = d.get("photo")
    used = [f"{fact}={name}" for fact, name in (d.get("keys") or {}).items()]
    if d.get("photo"):
        used.append(f"photo={d['photo'][-70:]}")
    rep["record_keys"] = (used + [k for k in (probe_keys or []) if not is_private_field(k)])[:60]
    return rep


# Fields on the page that are nobody's business off the Mac: never in a
# report, not even the probe's (the probe of 2026-09-14 sent the chassis
# number, the rule is never capture a VIN).
_PRIVATE_FIELD_RE = re.compile(r"chassis|vin\b|engine_number|customer|enquiry|mot_number|payment|finance|purchase|cost|price|vat|dealer_id|seller|supplier|stock_plan|reference", re.I)


def is_private_field(key):
    return bool(_PRIVATE_FIELD_RE.search(key or ""))
