"""
Auction4Cars reader.

Reads the logged in Auction4Cars live auctions list at auction4cars.com and
turns each card into a Car. The list card gives make, model, derivative,
year, mileage, transmission, fuel, location and price, but not the
registration, owners, or a guide value, those are only on each car's own
page, so every candidate needs a detail visit (the same shape as Carwow).

The real reserve IS available (found 2026-08-20), just never rendered as
visible text: each detail page carries priceCon.data('reserve', N) in an
inline script, read by _real_reserve(). Guide Clean (the platform's own
lower guide value) is kept as a fallback for the rare page that script is
not found on. This corrects the original 2026-08-19 assumption that no
reserve is ever disclosed on this platform, it just took reading the page
source to find it.

This platform still has no 1 to 5 condition grade, only a defect map and
tyre depths (Mark 2026-08-19). The grade gate is skipped for this source
(pricing.assess(..., assume_grade_ok=True)), condition is judged by eye
from the photos on the card instead.
Distance is a depot name (for example A4C-Castleford), not a mile figure, so
the distance gate is skipped too (assume_distance_ok=True) and the depot name
is carried in car.location for display.

Nothing here bids. It reads the screen only.
"""

import re
import datetime
import html as htmllib
from ..pricing import Car
from .. import browser

MONEY = re.compile(r"£\s*([\d,]+)")
YEAR = re.compile(r"^(19[89]\d|20[0-4]\d)$")

# Multi word makes, matched longest first so "Land Rover" is not read as make
# "Land" model "Rover". Everything else defaults to first word is the make.
_MULTI_WORD_MAKES = [
    "LAND ROVER", "MERCEDES-BENZ", "MERCEDES BENZ", "ALFA ROMEO",
    "ASTON MARTIN", "MG MOTOR UK", "GREAT WALL",
]



def _num(s):
    if s is None:
        return None
    s = str(s).replace(",", "").strip()
    return int(s) if s.isdigit() else None


def _money(s):
    if not s:
        return None
    m = MONEY.search(s)
    return _num(m.group(1)) if m else None


def _title_case(s):
    # Card titles are shouty caps ("BMW 3 SERIES"). Store makes and models the
    # way the platform's own "Browse Auctions by Make" list writes them.
    return " ".join(w.capitalize() if not w.isupper() or len(w) > 3 else w for w in s.split())


def _split_make_model(title):
    up = (title or "").strip().upper()
    for mk in _MULTI_WORD_MAKES:
        if up.startswith(mk):
            rest = title.strip()[len(mk):].strip()
            return _title_case(mk), _title_case(rest)
    parts = title.strip().split(" ", 1)
    make = parts[0] if parts else ""
    model = parts[1] if len(parts) > 1 else ""
    return _title_case(make), _title_case(model)


def _card_blocks(page_html):
    return page_html.split('<div class="vehicle-card-container">')[1:]


def parse_listing(page_html, today=None, days_ahead=7):
    """Parse one page of the live auctions list into partial Cars (no reg,
    owners or guide value yet, those need enrich_from_detail). Every card on
    this page is a live auction, there is no other listing type mixed in.

    Unlike Motorway and Carwow, an Auction4Cars listing does not sell the next
    day, auctions run 3, 5 or 7 days (Mark 2026-08-19). Originally this kept
    only auctions ending today or tomorrow, matching how far ahead Motorway
    and Carwow ever look; widened 2026-08-21 (Mark: add a cockpit filter for
    how many days ahead to show, which needs the days actually captured to
    filter across) to the full days_ahead window, default 7, the longest a
    real Auction4Cars auction runs, so nothing is dropped before the cockpit
    even gets a chance to filter it. A car whose auction-end-date is further
    out than that is still silently skipped here, the same way Carwow
    silently skips a non auction listing state. today defaults to the real
    date, pass it explicitly to keep a run's cars consistent with its
    sale_date."""
    today = today or datetime.date.today()
    # A proper inclusive range, not a two date membership test: the old
    # today or tomorrow only version got away with checking membership of a
    # 2 element tuple because there were only ever two valid dates, that
    # trick breaks the moment the window is wider than one day.
    window_end = today + datetime.timedelta(days=days_ahead)
    cars = []
    for block in _card_blocks(page_html):
        aid = re.search(r'data-auctionid="(\d+)"', block)
        auction_id = aid.group(1) if aid else ""
        if not auction_id:
            continue

        end = re.search(r'data-auction-end-date="(\d+)"', block)
        if not end:
            continue  # no end time read, cannot confirm it is in the window
        ends_at = datetime.datetime.fromtimestamp(int(end.group(1)) / 1000)
        if not (today <= ends_at.date() <= window_end):
            continue

        title = re.search(r'<h3 class="title">([^<]*)</h3>', block)
        title = htmllib.unescape(title.group(1)).strip() if title else ""
        make, model = _split_make_model(title)

        sub = re.search(r'<p class="sub-title"[^>]*title="([^"]*)"', block)
        derivative = htmllib.unescape(sub.group(1)).strip() if sub else ""

        loc = re.search(r'<div class="location">\s*<span class="icon-location"></span>\s*<span>([^<]*)</span>',
                        block)
        location = htmllib.unescape(loc.group(1)).strip() if loc else ""

        details = re.search(r'<div class="details">([^<]*)</div>', block)
        year = mileage = None
        transmission = fuel = ""
        if details:
            parts = [p.strip() for p in htmllib.unescape(details.group(1)).split("|")]
            for p in parts:
                if YEAR.match(p):
                    year = _num(p)
                elif "mile" in p.lower():
                    mileage = _num(re.sub(r"\D", "", p))
                elif p.lower() in ("manual", "automatic", "semi-automatic", "semi automatic"):
                    transmission = p
                elif p.lower() in ("petrol", "diesel", "hybrid", "electric"):
                    fuel = p

        price = re.search(r'<span class="price">([^<]*)</span>', block)
        price = _money(price.group(1)) if price else None

        photo = re.search(r'<img class="lozad" data-src="([^"]*)"', block)
        photo_url = photo.group(1) if photo else ""

        cars.append(Car(
            make=make, model=model, derivative=derivative,
            year=year, mileage=mileage, transmission=transmission, fuel=fuel,
            # Set at read time, not only in enrich_from_detail (Steven
            # 2026-09-02, Step 2 of the review): a car that fails the cheap
            # list gate is never visited, so with no engine text it was
            # rejected as "engine could not be confirmed" instead of its
            # real reason (age, mileage), 5,000 misattributed rows in a
            # fortnight. Same derivative plus fuel shape Carwow uses.
            engine=(derivative + (" " + fuel if fuel else "")).strip(),
            location=location, reserve=price, photo_url=photo_url,
            auction_ends_at=ends_at.isoformat(),
            listing_url=f"https://www.auction4cars.com/VehicleAuction/AuctionPage/{auction_id}",
            source="Auction4Cars",
        ))
    return cars


def _detail_fields(page_html):
    """Every label: value pair in the Vehicle Details block, keyed by the
    label text with the trailing colon and non breaking space stripped."""
    out = {}
    for m in re.finditer(
        r'<div class="car-detail-title">([^<]*)</div>\s*<div class="car-detail-content">\s*([^<]*?)\s*</div>',
        page_html,
    ):
        label = htmllib.unescape(m.group(1)).replace("\xa0", " ").strip().rstrip(":").strip()
        value = htmllib.unescape(m.group(2)).strip()
        out[label] = value
    return out


def _service_history(page_html):
    types = re.findall(r"<td>\s*[\d/]+\s*</td>\s*<td>\s*[\d,]*\s*</td>\s*<td>\s*([^<]*?)\s*</td>",
                       page_html)
    types = [t.strip() for t in types if t.strip()]
    if not types:
        return ""
    return "full" if all(t.lower() == "main dealer" for t in types) else "partial"


_RESERVE_RE = re.compile(r"priceCon\.data\('reserve',\s*([\d.]+)\)")


def _real_reserve(page_html):
    """The real reserve, found in an inline script on the car's own detail
    page (priceCon.data('reserve', 5000.0000);), not shown as visible text
    anywhere on the page (Mark 2026-08-20, found by reading the page source).
    Corrects the earlier assumption that Auction4Cars never discloses a
    reserve, it does, it is just never rendered. Returns None if the pattern
    is not found, never a guess."""
    m = _RESERVE_RE.search(page_html)
    if not m:
        return None
    try:
        return round(float(m.group(1)))
    except ValueError:
        return None


def _notes(page_html):
    m = re.search(r'<h3>More Information</h3>(.*?)<div class="clearfix">', page_html, re.S)
    if not m:
        return ""
    paras = re.findall(r"<p[^>]*>([^<]*)</p>", m.group(1))
    lines = [htmllib.unescape(p).replace("\xa0", "").strip() for p in paras]
    lines = [l for l in lines if l and "SVA" not in l]
    return " ".join(lines)


def enrich_from_detail(page, car):
    """Visit the car's own page for the reg, owners, real reserve and service
    history the list card does not carry. The real reserve is read from an
    inline script (priceCon.data('reserve', ...), Mark 2026-08-20), not
    visible as text anywhere on the page, which is why it was originally
    thought Auction4Cars never discloses one. Guide Clean is kept as a
    fallback for the rare page where that script is not found. The grade
    gate is still skipped entirely for this source, so no grade is read
    here."""
    page.goto(car.listing_url, wait_until="domcontentloaded", timeout=30000)
    page.wait_for_selector("#car-details-container", timeout=20000)
    html = page.content()

    fields = _detail_fields(html)
    car.reg = fields.get("Reg Number", "").strip()
    car.owners = _num(fields.get("Owners"))
    car.reserve = _real_reserve(html) or _money(fields.get("Guide Clean")) or car.reserve
    car.service_history = _service_history(html)
    car.notes = _notes(html)
    if not car.transmission:
        car.transmission = fields.get("Transmission", "")
    if not car.fuel:
        car.fuel = fields.get("Fuel", "")
    car.engine = (car.derivative + (" " + car.fuel if car.fuel else "")).strip()

    # Just the one photo per car (Mark 2026-08-24: "pull one image of each
    # car see if this can complete the scrape faster"), cut back from the
    # full gallery captured 2026-08-21. Every photo lives at
    # /AuctionImages/{auction id}/Image_N.jpg, a plain sequential pattern
    # with nothing document like mixed in, so this is inherently safe to
    # match, just kept to one now.
    if not car.photo_urls:
        car.photo_urls = _gallery_photos_from_html(html)[:1]

    return car


_GALLERY_PHOTO_RE = re.compile(r"https://www\.auction4cars\.com/AuctionImages/\d+/Image_\d+\.jpg")


def _gallery_photos_from_html(html):
    """The full listing gallery from a car's own detail page HTML, deduped
    to one URL per distinct photo. A standalone function purely for
    testing, enrich_from_detail is the only real caller."""
    seen = {}
    for gm in _GALLERY_PHOTO_RE.finditer(html):
        seen.setdefault(gm.group(0), gm.group(0))
    return list(seen.values())


def read_all_pages(page, today=None, days_ahead=7, max_pages=None, on_page=None):
    """Paginate the live auctions list from an already open, already
    navigated page (the page dropdown at the foot of the list jumps straight
    to any page number, proven live: selecting page 2 loads a wholly
    different set of cars), keeping only cars whose auction ends within
    days_ahead days. The caller owns the page and its context, so the daily
    run can reuse the same page afterward for detail enrichment without
    opening a second session.

    The list's own default sort is soonest ending first, so once two pages in
    a row have nothing in the window everything after is later still, and
    reading stops early rather than working through all 31 or so pages. This
    is a speed optimisation only, not a correctness one: parse_listing always
    filters by the real end date regardless of sort order, so a wrong
    assumption here means a slower read, never a wrongly included car."""
    cars = parse_listing(page.content(), today=today, days_ahead=days_ahead)
    sel = page.query_selector("div.pagination-container select.form-select")
    total_pages = 1
    if sel:
        opts = sel.query_selector_all("option")
        values = [o.get_attribute("value") for o in opts if o.get_attribute("value")]
        if values:
            total_pages = max(int(v) for v in values)
    if max_pages:
        total_pages = min(total_pages, max_pages)

    empty_pages = 0
    for n in range(2, total_pages + 1):
        sel = page.query_selector("div.pagination-container select.form-select")
        if not sel:
            break
        sel.select_option(str(n))
        page.wait_for_timeout(1800)
        found = parse_listing(page.content(), today=today, days_ahead=days_ahead)
        cars.extend(found)
        if on_page:
            on_page(n, len(cars))
        empty_pages = 0 if found else empty_pages + 1
        if empty_pages >= 2:
            break
    return cars


def read_live(playwright, headless=True, today=None, days_ahead=7, max_pages=None):
    """Standalone convenience: opens its own context, reads the live auctions
    (Cars tab) list for cars ending within days_ahead days, and closes.
    Returns partial Cars, reg, owners and the guide value are not read yet,
    that needs enrich_from_detail per car. The daily run does not use this,
    it manages its own context so the same page can be reused for
    enrichment, see daily_run._read_auction4cars. Fails loudly if the list
    does not render."""
    ctx = browser.open_reader_context(playwright, "auction4cars", headless=headless)
    try:
        page = ctx.pages[0] if ctx.pages else ctx.new_page()
        page.goto(browser.SITES["auction4cars"]["stock_url"], wait_until="domcontentloaded", timeout=45000)
        page.wait_for_selector('div.vehicle-card-container', timeout=30000)
        page.wait_for_timeout(1500)
        # A structural check separate from the date window: real cards render
        # (proven by the selector above), so if none of them even carry a
        # readable auction id the page itself has likely changed. An empty
        # result after date filtering is a legitimate, expected outcome (no
        # auctions happen to end within the window) and must not raise.
        if not re.search(r'data-auctionid="\d+"', page.content()):
            raise RuntimeError("Auction4Cars list rendered but no cars were parsed. The page may have changed.")
        return read_all_pages(page, today=today, days_ahead=days_ahead, max_pages=max_pages)
    finally:
        ctx.close()


# The account's own real purchase history (Mark 2026-08-27: "i now want to
# add a4c purchase info in to BB"), found live under My Account: "Won" only
# ever covers the last 3 months and, for this account, reads 0 (nothing
# currently mid process), while History goes back 13 months and holds every
# completed purchase, so History is the real source, the same "the wider,
# completed record, not the narrower in-flight one" choice already made for
# Motorway (Purchases > Complete) and Carwow (the won screen's own 90 day
# window). No registration or mileage is on this list page, only on each
# car's own detail page, the same "list card omits it, detail has it" shape
# already true for a live Auction4Cars listing (see enrich_from_detail
# above), reused directly here via _detail_fields, identical markup.
HISTORY_URL = "https://www.auction4cars.com/DealerAdmin/MyAccount/History"

_MONTHS_FULL = {m.lower(): i for i, m in enumerate(
    ["January", "February", "March", "April", "May", "June", "July",
     "August", "September", "October", "November", "December"], 1)}
_DATE_WON_RE = re.compile(r"^(\d{1,2})\s+([A-Za-z]+)\s+(\d{4})$")


def _parse_history_date(text):
    """"28 January 2026" (the History page's own real format, a full month
    name, never abbreviated the way Motorway's own purchases screen is)
    turned into a plain YYYY-MM-DD string. Returns "" rather than guessing
    if the text does not match, never seen any other shape live."""
    m = _DATE_WON_RE.match((text or "").strip())
    if not m:
        return ""
    day, month_name, year = int(m.group(1)), m.group(2).lower(), int(m.group(3))
    month = _MONTHS_FULL.get(month_name)
    if not month:
        return ""
    try:
        return datetime.date(year, month, day).isoformat()
    except ValueError:
        return ""


def parse_history(page_html):
    """Parse the My Purchase History list into partial purchase dicts, one
    per <div class="bnt-container bnt-history">: the auction id, vehicle
    name, purchase price (this figure already includes buyer's fees, the
    same real "what was actually paid" basis Motorway's and Carwow's own
    purchase price fields already use, confirmed live against a real
    auction: the detail page's own "Final Price" reads lower, the raw
    winning bid before fees) and the date won. No reg or mileage yet, that
    needs a detail page visit per row, see read_purchase_history."""
    out = []
    for block in page_html.split('<div class="bnt-container bnt-history">')[1:]:
        aid = re.search(r'href="/VehicleAuction/AuctionPage/(\d+)"', block)
        if not aid:
            continue
        auction_id = aid.group(1)
        name = re.search(r'class="auctionResult-table-vehicleName">([^<]*)</a>', block)
        name = htmllib.unescape(name.group(1)).strip() if name else ""
        price = re.search(r"Purchase Price:\s*</div>\s*<div>\s*£([\d,]+)\s*</div>", block)
        price = _num(price.group(1)) if price else None
        won = re.search(r"Date Won:\s*</div>\s*<div>\s*([^<]*?)\s*</div>", block)
        bought_date = _parse_history_date(won.group(1)) if won else ""
        if not name:
            continue
        out.append({
            "platform": "Auction4Cars", "vehicle_ref": auction_id, "reg": "",
            "name": name, "status": "", "price": price, "bought_date": bought_date,
            "listing_url": f"https://www.auction4cars.com/VehicleAuction/AuctionPage/{auction_id}",
        })
    return out


def read_purchase_history(playwright, headless=True):
    """Read the account's own real purchase history: every car ever bought
    on Auction4Cars, newest first, over the platform's own 13 month window.
    One continuous session, the list read once then every entry's own
    detail page visited in turn for its real registration and mileage
    (never on the list page itself), the same shape purchases_run.py's own
    Motorway and Carwow blocks already use for their own per car detail
    reads. Fails loudly if the History screen itself does not render; one
    bad detail page is skipped rather than failing the whole read, its own
    reg is simply left blank. Read only, never acts."""
    ctx = browser.open_reader_context(playwright, "auction4cars", headless=headless)
    try:
        page = ctx.pages[0] if ctx.pages else ctx.new_page()
        page.goto(HISTORY_URL, wait_until="domcontentloaded", timeout=45000)
        try:
            page.wait_for_selector(".auctionResult-table, .tab-pane", timeout=30000)
        except Exception:
            raise RuntimeError(
                "Auction4Cars purchase history screen did not render. The page may "
                "have changed or the login may have expired. "
                "Run: python3 login.py auction4cars manual")
        rows = parse_history(page.content())
        for r in rows:
            try:
                page.goto(r["listing_url"], wait_until="domcontentloaded", timeout=30000)
                page.wait_for_selector("#car-details-container", timeout=20000)
                fields = _detail_fields(page.content())
                r["reg"] = fields.get("Reg Number", "").strip()
                r["mileage"] = _num(fields.get("Mileage"))
            except Exception as e:
                print(f"Auction4Cars purchase detail read failed for auction {r['vehicle_ref']}: {e}")
        return rows
    finally:
        ctx.close()


def deep_read(page, listing_url, asked_at=None):
    """The shortlist deep read of one Auction4Cars car page (2026-09-17):
    its HTML (the Vehicle Details block, the service table, the reserve
    script, every photo), handed to bidbrain.deep_read.from_auction4cars."""
    from bidbrain import deep_read as _deep
    page.goto(listing_url, wait_until="domcontentloaded", timeout=30000)
    try:
        page.wait_for_selector("#car-details-container", timeout=20000)
    except Exception:
        return _deep.failed("Auction4Cars' car page did not load; the login may have expired. Run: python3 login.py auction4cars manual", asked_at)
    d = _deep.from_auction4cars(page.content())
    d["read_at"] = _deep.now_stamp()
    d["asked_at"] = asked_at
    return d
