"""
Dealer Auction reader (Auto Trader's trade auction platform, added 2026-08-20).

Reads the logged in "Dealer Auction" tab of the dealer sourcing portal at
dealerauction.co.uk. Unlike Motorway and Carwow, almost everything needed is
already on the list card itself, registration, grade, owners, VAT status,
retail value and margin, and a time remaining countdown, so there is very
little detail page dependency here.

Three things decided by Mark 2026-08-20:
- No reserve is ever disclosed anywhere on this platform (checked the list
  and a real detail page thoroughly, unlike Auction4Cars, which turned out to
  have one hidden). Retail minus Margin stands in for it instead, labelled
  "AT Trade" on the card (the platform's own two real figures, not a guess).
- Condition grade is the real NAMA 1 to 5 scale, shown directly on the card,
  so this source needs no grade gate exemption, unlike Auction4Cars.
- Distance is never exact miles, only a town plus county string. Mark chose
  a county level allow list (like Auction4Cars' depot list, but coarser,
  county rather than town, since town names would be too many to manage).
  See pricing.gate_failures and _dealerauction_county.

Nothing here bids. It reads the screen only.
"""

import re
import datetime
import html as htmllib
from .. import browser
from ..pricing import Car

MONEY = re.compile(r"£\s*([\d,]+)")
YEAR = re.compile(r"^(19[89]\d|20[0-4]\d)$")
# A "recommended"/"similar" card carries a ?referrer=... suffix on its own
# id before the closing quote, still a real distinct listing, not noise.
ADVERT_ID_RE = re.compile(r'href="#/advert/([A-Za-z0-9]+)(?:\?[^"]*)?"')
NEXT_SEL = 'button[analyticslabel="paginator-next"]'

# Multi word makes, matched longest first so "Land Rover" is not read as make
# "Land" model "Rover". Same reasoning and shape as auction4cars.py's list.
_MULTI_WORD_MAKES = [
    "LAND ROVER", "MERCEDES-BENZ", "MERCEDES BENZ", "ALFA ROMEO",
    "ASTON MARTIN", "GREAT WALL",
]

# The account's own sourcing view, all networks Auto Trader bundles under the
# "Dealer Auction" tab (captured live 2026-08-20, this is the tab's own
# default selection, separate from "Private AT listings" and the fully
# separate Manheim tab), filtered to auction only (salesMethod=bid). Buy it
# now and Make me an offer listings are excluded by this same query param,
# matching the auction only rule already applied to every other platform.
STOCK_URL = (
    "https://www.dealerauction.co.uk/sourcing/#/search"
    "?network=trade_independent&network=trade_franchise&network=trade_fleet"
    "&network=retail_ready&network=c2b&network=c2b_generic"
    "&network=digital_wholesale&network=hertz&network=manheim_digital_first"
    "&salesMethod=bid"
)


def _num(s):
    if s is None:
        return None
    s = str(s).replace(",", "").strip()
    return int(s) if s.isdigit() else None


def _money(s):
    if not s:
        return None
    m = MONEY.search(s)
    return _num(m.group(1)) if m else None


def _title_case(s):
    return " ".join(w if (w.isupper() and len(w) <= 3) else w.capitalize() for w in s.split())


def _split_make_model(title):
    """MAKE and MODEL are both in shouty caps in the title text, the
    derivative is not, for example "LAND ROVER RANGE ROVER SPORT 2.0 P400e
    HSE Dynamic..." splits as make "Land Rover", model "Range Rover Sport",
    derivative "2.0 P400e...". The boundary is the first token that either
    contains a lowercase letter or looks like an engine size ("1.7", "2.0"),
    everything before that is still MAKE plus MODEL. This is a good enough
    split for display and filtering, not perfect on every trim naming
    (matches the same accepted imprecision as auction4cars.py's own split,
    it is never used for the hard gate itself, only banned make matching,
    which only ever needs the MAKE half, always reliably identified here)."""
    title = (title or "").strip()
    tokens = title.split()
    if not tokens:
        return "", "", ""
    up = title.upper()
    make_tokens = None
    for mk in _MULTI_WORD_MAKES:
        mk_tokens = mk.split()
        if up.split()[:len(mk_tokens)] == mk_tokens:
            make_tokens = tokens[:len(mk_tokens)]
            rest = tokens[len(mk_tokens):]
            break
    if make_tokens is None:
        make_tokens = tokens[:1]
        rest = tokens[1:]

    model_tokens = []
    for t in rest:
        if re.match(r"^\d+\.\d", t) or any(c.islower() for c in t):
            break
        model_tokens.append(t)
    derivative_tokens = rest[len(model_tokens):]

    make = _title_case(" ".join(make_tokens))
    model = _title_case(" ".join(model_tokens)) if model_tokens else ""
    derivative = " ".join(derivative_tokens)
    return make, model, derivative


def _parse_time_remaining(text, now=None):
    """"10h 35m", "2d 3h", "45m" etc, added to now to give an absolute end
    time for the cockpit's live countdown (same mechanism Auction4Cars uses,
    car.auction_ends_at). Returns "" if the text does not parse, so a card
    with a reading that cannot be trusted just shows no countdown rather than
    a wrong one."""
    if not text:
        return ""
    days = re.search(r"(\d+)\s*d", text)
    hours = re.search(r"(\d+)\s*h", text)
    mins = re.search(r"(\d+)\s*m", text)
    if not (days or hours or mins):
        return ""
    now = now or datetime.datetime.now()
    delta = datetime.timedelta(
        days=int(days.group(1)) if days else 0,
        hours=int(hours.group(1)) if hours else 0,
        minutes=int(mins.group(1)) if mins else 0,
    )
    return (now + delta).isoformat()


def _card_blocks(page_html):
    """Splits cleanly one block per card, verified against a real captured
    page (37 parts for 36 real cards, the first part is the pre amble before
    any card). The earlier attempt split on a CSS class that turns out to
    repeat inside a card (a "similar listings" widget shares it), which
    silently dropped and truncated cards, caught before this shipped."""
    return page_html.split('href="#/advert/')[1:]


def parse_listing(page_html, now=None):
    """Parse one page of the Dealer Auction list into Cars. Registration,
    grade, owners, VAT status, retail value and margin are all already on the
    card, unlike Motorway and Carwow, so no detail page visit is needed for
    the gate itself."""
    cars = []
    ids_seen = set()
    for block in _card_blocks(page_html):
        # A card reached via a "recommended" or "similar" widget carries a
        # ?referrer=... suffix on its own id before the closing quote, still
        # a real distinct listing, not noise.
        id_m = re.match(r'([A-Za-z0-9]+)(?:\?[^"]*)?"', block)
        advert_id = id_m.group(1) if id_m else ""
        if not advert_id or advert_id in ids_seen:
            continue
        ids_seen.add(advert_id)

        title = re.search(r'title="([^"]+)"[^>]*>\s*([^<]*)</a>', block)
        title_text = htmllib.unescape(title.group(2)).strip() if title else ""
        make, model, derivative = _split_make_model(title_text)

        vrm = re.search(r'analyticslabel="copy-vrm"[^>]*>\s*([^<]+?)\s*<', block)
        reg = "".join((vrm.group(1) if vrm else "").split()).upper()

        year_m = re.search(r'test-advert-combinedYearPlate"[^>]*>\s*(\d{4})\s*\(\d+\)\s*<', block)
        year = _num(year_m.group(1)) if year_m else None

        body_m = re.search(r'test-advert-bodyType"[^>]*>\s*([A-Za-z]+)\s*<', block)
        body_type = body_m.group(1) if body_m else ""

        mileage_m = re.search(r"([\d,]+)\s*miles", block)
        mileage = _num(mileage_m.group(1).replace(",", "")) if mileage_m else None

        trans_m = re.search(r">\s*(Manual|Automatic|Semi-automatic|Semi automatic)\s*<", block)
        transmission = trans_m.group(1) if trans_m else ""

        fuel_m = re.search(r">\s*(Petrol|Diesel|Hybrid|Electric|Petrol Plug-in Hybrid|Petrol Hybrid)\s*<", block)
        fuel = fuel_m.group(1) if fuel_m else ""

        owners_m = re.search(r"(\d+)\s*previous owners?", block)
        owners = _num(owners_m.group(1)) if owners_m else None

        vat_m = re.search(r'test-advert-vatStatus"[^>]*>\s*VAT (\w+)\s*<', block)
        vat_qualifying = bool(vat_m and vat_m.group(1) == "Qualifying")

        grade_m = re.search(r"Condition grade (\d)", block)
        grade = _num(grade_m.group(1)) if grade_m else None

        # The title attribute carries clean "Town, County" capitalisation,
        # the visible inner text is lower case with stray extra spaces.
        loc_m = re.search(r'test-location-info"\s+title="([^"]+)"', block)
        location = htmllib.unescape(loc_m.group(1)).strip() if loc_m else ""

        retail_m = re.search(r">\s*Retail\s*<.{0,200}?£([\d,]+)", block, re.S)
        retail = _num(retail_m.group(1)) if retail_m else None
        margin_m = re.search(r">\s*Margin\s*<.{0,200}?£([\d,]+)", block, re.S)
        margin = _num(margin_m.group(1)) if margin_m else None
        # Mark 2026-08-20: no real reserve is ever disclosed on this
        # platform, Retail minus Margin (the platform's own two figures)
        # stands in for it, labelled "AT Trade" on the card, never a guess
        # since both halves are read, not estimated.
        reserve = (retail - margin) if (retail is not None and margin is not None) else None

        time_m = re.search(r"Time remaining.{0,400}?<strong[^>]*>([^<]*)</strong>", block, re.S)
        ends_at = _parse_time_remaining(time_m.group(1) if time_m else "", now=now)

        # Every card carries several small images alongside the real vehicle
        # photo (a network source badge like /img/logos/manheim.png, an
        # inspection report badge), and on some cards those render before
        # the gallery in the raw markup, so a bare first src/data-src match
        # picked up a badge icon instead of the car (Mark 2026-08-21, most
        # cards showing a broken image). The gallery's own image is
        # reliably marked alt="Primary vehicle image", checked per <img> tag
        # rather than assuming attribute order, so this only ever reads the
        # real photo, never a badge.
        photo_url = ""
        for img_tag in re.findall(r"<img\b[^>]*>", block):
            if 'alt="Primary vehicle image"' in img_tag:
                src_m = re.search(r'src="([^"]+)"', img_tag)
                if src_m:
                    photo_url = src_m.group(1)
                break

        # Retail here is Auto Trader's own retail valuation, a different
        # scale to CAP Clean (used elsewhere for the value caps and the
        # "looks high against CAP" sanity flag), so it is deliberately not
        # stored as car.cap_clean, that would silently apply those checks
        # against the wrong kind of figure. Those checks simply do not fire
        # for this platform, same as any other car with no CAP reading.
        # The derivative text carries the real engine identifier ("1.0
        # EcoBoost", "1.7 CRDi", "2.0 TD4"...), same convention as
        # auction4cars.py: engine is derivative plus fuel, used for the
        # banned engine and litre ban checks. Without this every car
        # rejected as "engine could not be confirmed", caught live before
        # this shipped (all 116 real list gate passers were rejecting).
        engine = (derivative + (" " + fuel if fuel else "")).strip()

        cars.append(Car(
            reg=reg, make=make, model=model, derivative=derivative,
            year=year, mileage=mileage, transmission=transmission, fuel=fuel,
            engine=engine, body_type=body_type, owners=owners, grade=grade,
            vat_qualifying=vat_qualifying, reserve=reserve,
            location=location, photo_url=photo_url,
            auction_ends_at=ends_at,
            listing_url=f"https://www.dealerauction.co.uk/sourcing/#/advert/{advert_id}",
            source="Dealer Auction",
        ))
    return cars


def read_all_pages(page, max_pages=60, on_page=None):
    """Accumulate cars across every page of the already filtered (auction
    only) list, from an already open, already logged in page. Stops when a
    page repeats the same advert ids as the one before, or after max_pages as
    a safety cap. A "Recommended Stock" carousel on the page reuses some of
    the same cars across different page loads, so each page's cars are kept
    only if their registration has not already been added, not just added in
    bulk (a real duplicate car bug caught live, about a third of a naive
    accumulation turned out to be repeats from that carousel)."""
    cars = []
    seen_ids = set()
    seen_regs = set()
    for i in range(max_pages):
        html = page.content()
        page_ids = set(ADVERT_ID_RE.findall(html))
        new_ids = page_ids - seen_ids
        if not new_ids and seen_ids:
            break
        seen_ids |= page_ids
        for c in parse_listing(html):
            if c.reg and c.reg in seen_regs:
                continue
            if c.reg:
                seen_regs.add(c.reg)
            cars.append(c)
        if on_page:
            on_page(i + 1, len(cars))

        nxt = page.query_selector(NEXT_SEL)
        if not nxt:
            break
        try:
            nxt.click(timeout=3000)
        except Exception:
            break
        page.wait_for_timeout(1800)
    return cars


def read_live(playwright, headless=True, max_pages=60):
    """Standalone convenience wrapper: opens its own context, reads the whole
    auction only list, closes. Fails loudly if the list never renders."""
    ctx = browser.open_reader_context(playwright, "dealerauction", headless=headless)
    try:
        page = ctx.pages[0] if ctx.pages else ctx.new_page()
        page.goto(STOCK_URL, wait_until="networkidle", timeout=30000)
        try:
            page.wait_for_selector('a[analyticslabel="view-advert"]', timeout=20000)
        except Exception:
            raise RuntimeError(
                "Dealer Auction list did not render any vehicles "
                f"(no advert links at {page.url}). The page may have changed "
                "or the login may have expired. Run: python3 login.py dealerauction manual"
            )
        page.wait_for_timeout(800)
        cars = read_all_pages(page, max_pages=max_pages)
        if not cars:
            raise RuntimeError(
                "Dealer Auction list rendered but no vehicles were read. "
                "The layout may have changed, not guessing."
            )
        return cars
    finally:
        ctx.close()


# The account's own real purchase record (Mark 2026-08-27: "ok lets do DA
# now", the same "add purchase info in to BB" ask already done for
# Auction4Cars). Found live under Dashboard > Invoice history, a real
# itemised transaction ledger (a genuine "Purchases" filter tab exists
# alongside "All"/"Sales", though live checked both read identically for
# this account, every row already IS a purchase, still selected explicitly
# rather than relied on to stay that way). Unlike Auction4Cars, no detail
# page visit is needed at all: registration, the full real fee breakdown
# (vehicle price, buyer fee, VAT) and the advert id are all already on this
# one list, the richest of the three platforms' own purchase records built
# so far. Newest first, confirmed live, so a daily top up only needs the
# first page or two; a real pagination widget exists (analyticslabel=
# "paginator-next", the exact same one the stock list's own read_all_pages
# already drives, real live account showed 12 pages of 10 rows each).
INVOICE_HISTORY_URL = "https://www.dealerauction.co.uk/sourcing/#/invoice-history"
PURCHASES_TAB_SEL = 'button[analyticslabel="invoice-history-purchases"]'
_REG_LEAD_RE = re.compile(r"^([A-Z]{2}\d{2}\s?[A-Z]{3})\b")
_INVOICE_DATE_RE = re.compile(r"^(\d{2})/(\d{2})/(\d{2}),?\s+(\d{2}):(\d{2}):(\d{2})$")


def _parse_invoice_date(text):
    """"29/07/26, 14:44:39" (the invoice history page's own real format, a
    2 digit year) turned into a plain YYYY-MM-DD string. Returns "" rather
    than guessing if the text does not match, never seen any other shape
    live."""
    m = _INVOICE_DATE_RE.match((text or "").strip())
    if not m:
        return ""
    day, month, year2 = int(m.group(1)), int(m.group(2)), int(m.group(3))
    try:
        return datetime.date(2000 + year2, month, day).isoformat()
    except ValueError:
        return ""


def parse_invoice_history(page_html):
    """Parse one page of the Invoice history table into purchase dicts. Each
    real data row carries a class="c-recent-transactions__charge-id" cell
    (the header row does not, so it is skipped for free by requiring that
    marker). Once split on that marker, each row's own remaining columns
    (Status, Description, Date/Time, Vehicle price, Fee, Tax, Total, Type,
    Payment Method) are still in a fixed, real order, but the same generic
    "<td>...(?:(?!</td>).)*...</td>" trick used elsewhere in this project
    cannot isolate the Description column here: this table's own Status
    column ALSO carries a "<div class=...>Complete</div>" plus several
    Angular comment placeholders that make it look like real content to a
    naive first-td match (found live, the description regex kept matching
    the Status cell instead). Splitting on every real "</td><td>" boundary
    instead gives one clean chunk per column, addressed by its own fixed
    position, immune to that confusion. The registration sits at the very
    start of the Description column's own text ("DN18NCD Hyundai Santa
    Fe...", confirmed on every real row seen live), the rest up to " Buyer
    Fee" is the vehicle name, and the advert id is the real "View advert"
    link inside that same column (reusing ADVERT_ID_RE, the exact same
    pattern the stock list's own parse_listing already matches). Vehicle
    price, Fee and Tax are kept under the same generic winning_bid/
    motorway_fee/vat_total column names Carwow's own purchases already
    reuse (Mark's original platform specific naming, kept for db.py's own
    _PURCHASE_PAYMENT_FIELDS shape, no transport_fee or protect_fee
    equivalent shown on this platform)."""
    out = []
    for block in page_html.split('class="c-recent-transactions__charge-id"')[1:]:
        ref = re.search(r"<span>([^<]*)</span>", block)
        reference = htmllib.unescape(ref.group(1)).strip() if ref else ""
        if not reference:
            continue
        cols = re.split(r"</td>\s*<td[^>]*>", block)
        # cols[0] is the tail of this cell (the reference span plus Angular
        # comment placeholders); the real columns follow in order.
        if len(cols) < 9:
            continue
        status = re.search(r'c-pill[^"]*">([^<]*)</div>', cols[1])
        status = htmllib.unescape(status.group(1)).strip() if status else ""
        desc_html = cols[2]
        aid = ADVERT_ID_RE.search(desc_html)
        advert_id = aid.group(1) if aid else ""
        desc_text = htmllib.unescape(re.sub(r"<[^>]+>", " ", desc_html)).strip()
        desc_text = re.sub(r"\s+", " ", desc_text)
        reg_m = _REG_LEAD_RE.match(desc_text)
        reg = "".join(reg_m.group(1).split()) if reg_m else ""
        name = desc_text[reg_m.end():].strip() if reg_m else desc_text
        name = re.sub(r"\s*Buyer Fee\s*View advert\s*$", "", name).strip()
        bought_date = _parse_invoice_date(cols[3].strip())
        vehicle_price = _col_money(cols[4])
        fee = _col_money(cols[5])
        tax = _col_money(cols[6])
        total = _col_money(cols[7])
        if not reg or vehicle_price is None:
            continue
        out.append({
            "platform": "Dealer Auction", "vehicle_ref": reference, "reg": reg,
            "name": name, "status": status, "price": vehicle_price,
            "total_price": total, "bought_date": bought_date,
            "winning_bid": vehicle_price, "motorway_fee": fee, "vat_total": tax,
            "listing_url": (f"https://www.dealerauction.co.uk/sourcing/#/advert/{advert_id}"
                             if advert_id else ""),
        })
    return out


def _pence_money(digits):
    """A bare "212.50" (already extracted, the £ sign stripped by the
    caller's own regex) to a plain float, real pence kept, unlike the
    stock list's own _money/_num above, which only ever need whole
    pounds; a real invoice total genuinely carries them (£212.50). Named
    distinctly from _money above so it can never shadow it."""
    try:
        return float(digits.replace(",", ""))
    except (ValueError, AttributeError):
        return None


def _col_money(col_html):
    """The £ figure inside one already isolated invoice history column
    (a <strong>£N.NN</strong> cell), or None if that column carries none."""
    m = re.search(r"£([\d,.]+)", col_html)
    return _pence_money(m.group(1)) if m else None


def read_all_invoice_pages(page, max_pages=2):
    """Accumulate purchase rows across the Invoice history table's own real
    pagination (the same paginator-next button and shape read_all_pages
    already drives for the stock list), newest first, confirmed live, so
    max_pages defaults to a small daily top up rather than the whole
    history (a real account showed 12 pages of 10 rows each, 2026-08-27).
    Stops early once a page repeats an already seen reference, or there is
    no next page left."""
    rows = []
    seen_refs = set()
    for i in range(max_pages):
        html = page.content()
        found = parse_invoice_history(html)
        new = [r for r in found if r["vehicle_ref"] not in seen_refs]
        if not new and seen_refs:
            break
        seen_refs |= {r["vehicle_ref"] for r in found}
        rows.extend(new)
        nxt = page.query_selector(NEXT_SEL)
        if not nxt:
            break
        try:
            nxt.click(timeout=3000)
        except Exception:
            break
        page.wait_for_timeout(1800)
    return rows


def read_purchase_invoices(playwright, headless=True, max_pages=2):
    """Read the account's own real Invoice history, Purchases only, newest
    first. Fails loudly if the screen itself does not render; a purely
    empty result after that (a genuinely quiet stretch with nothing bought)
    is not an error. Read only, never acts."""
    ctx = browser.open_reader_context(playwright, "dealerauction", headless=headless)
    try:
        page = ctx.pages[0] if ctx.pages else ctx.new_page()
        page.goto(INVOICE_HISTORY_URL, wait_until="networkidle", timeout=45000)
        try:
            page.wait_for_selector(PURCHASES_TAB_SEL, timeout=30000)
        except Exception:
            raise RuntimeError(
                "Dealer Auction invoice history screen did not render. The page may "
                "have changed or the login may have expired. "
                "Run: python3 login.py dealerauction manual")
        page.click(PURCHASES_TAB_SEL)
        page.wait_for_timeout(1500)
        return read_all_invoice_pages(page, max_pages=max_pages)
    finally:
        ctx.close()


# The advert's own image viewer (Mark 2026-08-27: "still no images in the DA
# cars", the Invoice history list carries none at all, checked live, no
# <img> tag anywhere on that page). Found live on a real won advert's own
# page (a client side route, #/advert/<id>, the exact same address every
# row's own listing_url already points at): each photo is
# class="c-fpa-image-viewer__image" alt="Main image N". The first dozen or
# so load at full "preview.jpeg" quality straight into src; the rest stay
# lazy (a low res "thumbnail.jpeg" in src) until scrolled into view, but
# always carry the real preview quality url in data-full even before that,
# so data-full is preferred over src whenever both are present.
_ADVERT_IMG_RE = re.compile(r'<img[^>]*class="c-fpa-image-viewer__image"[^>]*>')
_ADVERT_IMG_IDX_RE = re.compile(r'alt="Main image (\d+)"')
_ADVERT_IMG_SRC_RE = re.compile(r'src="([^"]+)"')
_ADVERT_IMG_FULL_RE = re.compile(r'data-full="([^"]+)"')
ADVERT_URL_RE = re.compile(r"/advert/([A-Za-z0-9]+)")


def _parse_advert_gallery(page_html):
    """Every real photo on a won car's own advert page, ordered by the
    image's own position index (never scrape order), deduped by index. []
    on a page with no gallery at all, never guessed."""
    seen = {}
    for tag in _ADVERT_IMG_RE.findall(page_html):
        idx_m = _ADVERT_IMG_IDX_RE.search(tag)
        if not idx_m:
            continue
        idx = int(idx_m.group(1))
        full_m = _ADVERT_IMG_FULL_RE.search(tag)
        src_m = _ADVERT_IMG_SRC_RE.search(tag)
        url = full_m.group(1) if full_m else (src_m.group(1) if src_m else "")
        if url:
            seen[idx] = url
    return [seen[i] for i in sorted(seen)]


def read_purchase_gallery(page, advert_id):
    """The full photo gallery from a won car's own advert page, the only
    place Dealer Auction ever shows real photos for a purchase (the
    Invoice history list itself carries none). The page is a client side
    route and the gallery only finishes rendering a moment after
    navigation, so this waits the same real 4s the live reconnaissance for
    this needed. Returns [] rather than guessing if the page never renders
    a gallery at all (a listing that has since come down, for example)."""
    page.goto(f"https://www.dealerauction.co.uk/sourcing/#/advert/{advert_id}",
              wait_until="domcontentloaded", timeout=45000)
    page.wait_for_timeout(4000)
    try:
        html = page.content()
    except Exception:
        return []
    return _parse_advert_gallery(html)
