#!/usr/bin/env python3
"""
Backfills real mileage, Cazana retail value, year, transmission and engine
onto historical lost bids and purchases, so the AI suggested bid can
compare a winning price to that specific car's own retail value and to
genuinely similar cars, rather than a flat comps figure (Mark 2026-08-24:
"one thing im not sure we have is the % of cazana retail price taken in to
account for the ai prediction... having this info should improve the
accuracy", then 2026-08-25: "as we bid on cars over the next month this
list is gonna populate with more data... it would make sense to populate
this from our database as much as possible" and "compare year,
transmission, engine size, mileage so the ai is comparing as similar cars
as it can").

Three phases, resumable independently (a re run just picks up wherever it
left off, already enriched rows are skipped):

  python3 bid_retail_enrich.py --fields [--limit N]
      Database only, no browser: fills in year, transmission and engine on
      any lost bid or purchase missing them. Prefers a real sighting on
      file for that reg (BidBrain's own daily reads already log every car
      it ever looked at, not just the ones it went on to bid on), then
      falls back to deriving the year from the registration's own UK
      plate age identifier and an automatic badge from the platform's own
      free text vehicle name, both purely algorithmic, no lookup needed,
      since sightings alone barely covers any of the real history yet.
      Cheap and fast, always worth running first.

  python3 bid_retail_enrich.py --mileage [--limit N]
      For any lost bid still missing mileage or transmission, or any
      Motorway purchase still missing transmission, checks its own
      closest sighting first (free, instant) and only falls back to
      visiting its still viewable Motorway vehicle page live when that
      is not enough, reading real mileage, transmission and engine size
      off its own "Vehicle details" block in the one visit (Mark
      2026-08-25: "any way to increase the amount of cars you can
      suggest bids on", once transmission became a hard gate). One
      continuous Motorway session for whatever the database could not
      already answer.

  python3 bid_retail_enrich.py --retail [--limit N]
      For every lost bid that now has a mileage but no Cazana retail yet,
      looks it up live (a fresh Cazana session per car, matching how
      every other Cazana lookup in this project already works). No
      database shortcut exists for this one, a real valuation always
      needs a real lookup.

All three default to only the model groups with at least 10 real comps
(--min-samples to change), the "start with the biggest groups" scope Mark
chose over a full backfill of all 931, given how long a full run of either
phase genuinely takes. --limit caps how many rows this run touches, handy
for a real, honest time check before committing to the rest.

Reads screens only. Never bids, never acts on the platforms.
"""

import sys

from playwright.sync_api import sync_playwright

from bidbrain import db, browser, bid_suggest, comp_share
from bidbrain.readers import motorway, cazana
import dealer_config


def _target_keys(min_samples):
    # Counts shared comps from another dealership too (Mark 2026-08-25:
    # "we are both happy for CompMatch to learn from both our data"), a
    # model group thin on this Mac's own history but well covered by the
    # other dealership's is still worth the real time this script spends
    # enriching it.
    all_comps = db.bid_comps_full() + comp_share.load_shared_comps(
        exclude_dealer=getattr(dealer_config, "DEALER_NAME", ""))
    pool = bid_suggest.build_pool(all_comps)
    return {k for k, v in pool.items() if len(v) >= min_samples}


def _derived_fields(r, before_date):
    """Whatever year/transmission/engine/grade can be filled in for one
    row, a real sighting on file first (the most precise, an actual read
    of that exact car, and the ONLY source for grade and engine, neither
    of which can be guessed from a reg or a name), then the
    registration's own age identifier for year and the platform's own
    free text name for an automatic badge (Mark 2026-08-25, "make sure
    the age and mileage are being considered... seem a bit off":
    sightings alone covers barely any of the real history yet, 5 of
    1038, while the reg and name are on every row from day one). Never
    guesses a year off an older or private plate shape, never guesses
    Manual off a name with no automatic badge on it, never guesses a
    grade or engine at all."""
    s = db.latest_sighting_for_reg(r["reg"], before_date=before_date) or {}
    year = s.get("year") or bid_suggest.year_from_reg(r.get("reg"))
    transmission = s.get("transmission") or bid_suggest.transmission_from_name(r.get("name"))
    engine = s.get("engine")
    grade = s.get("grade")
    return year, transmission, engine, grade


def run_fields(min_samples, limit):
    # Unlike --mileage and --retail (real, time costly browser lookups,
    # deliberately scoped to the biggest model groups), this phase costs
    # nothing: no browser, no network, just a database join and a couple
    # of regexes. So it runs over EVERY lost bid and purchase, not just
    # the "biggest groups" scope, a smaller model group benefits just as
    # much from knowing its own comps' age and transmission.
    lb_rows = db.lost_bids_missing_fields()
    p_rows = db.purchases_missing_fields()
    if limit:
        lb_rows, p_rows = lb_rows[:limit], p_rows[:limit]
    print(f"{len(lb_rows)} lost bids and {len(p_rows)} purchases need year/"
          f"transmission/engine/grade (every model group, this phase is free)")
    done = 0
    for r in lb_rows:
        year, transmission, engine, grade = _derived_fields(r, r.get("date_bid"))
        if year is None and transmission is None and engine is None and grade is None:
            continue
        db.set_lost_bid_fields(r["id"], year=year, transmission=transmission,
                                engine=engine, grade=grade)
        done += 1
    for r in p_rows:
        year, transmission, engine, grade = _derived_fields(r, r.get("bought_date"))
        if year is None and transmission is None and engine is None and grade is None:
            continue
        db.set_purchase_fields(r["id"], year=year, transmission=transmission,
                                engine=engine, grade=grade)
        done += 1
    print(f"Fields: {done} of {len(lb_rows) + len(p_rows)} filled in from a real sighting, "
          f"the reg's own age identifier, or an automatic badge in the name "
          f"({len(lb_rows) + len(p_rows) - done} genuinely have none of the three available)")


def run_mileage(min_samples, limit):
    """Real page visits for whatever mileage, transmission or engine size
    the database and its own free derivations could not already fill
    (Mark 2026-08-25, once transmission became a hard gate in CompMatch,
    manual and automatic never mixed: "any way to increase the amount of
    cars you can suggest bids on"). A vehicle's own page shows a real,
    confirmed "Vehicle details" block (transmission, engine size,
    checked live), the same genuine source as a fresh daily read, not a
    badge guessed from a listing name. Covers both lost bids (mileage,
    transmission, engine, in the one visit already made for mileage) and
    Motorway purchases (transmission, engine only, mileage is already
    known for a purchase from the time it was actually bought)."""
    keys = _target_keys(min_samples)
    by_id = {}
    for r in db.lost_bids_missing_mileage():
        if bid_suggest.model_key(r["name"]) in keys:
            by_id[r["id"]] = r
    for r in db.lost_bids_missing_fields():
        if r.get("transmission") is None and bid_suggest.model_key(r["name"]) in keys:
            by_id.setdefault(r["id"], r)
    rows = list(by_id.values())
    if limit:
        rows = rows[:limit]

    p_rows = [r for r in db.purchases_missing_fields()
              if r["platform"] == "Motorway" and r.get("transmission") is None
              and r.get("listing_url") and bid_suggest.model_key(r["name"]) in keys]
    if limit:
        p_rows = p_rows[:limit]

    print(f"{len(rows)} lost bids and {len(p_rows)} Motorway purchases need a real page "
          f"visit for mileage, transmission or engine size (model groups with >= {min_samples} comps)")
    if not rows and not p_rows:
        return

    done, from_db = 0, 0
    still_needed = []
    for r in rows:
        s = db.latest_sighting_for_reg(r["reg"], before_date=r.get("date_bid"))
        if s and s.get("mileage") and r.get("mileage") is None:
            db.set_lost_bid_mileage(r["id"], s["mileage"])
            from_db += 1
        if r.get("transmission") is not None and (r.get("mileage") is not None or (s and s.get("mileage"))):
            continue  # a sighting (or the row itself) already covers everything this row needed
        still_needed.append(r)
    print(f"  {from_db} filled in straight from a sighting already on file, "
          f"{len(still_needed)} lost bids and {len(p_rows)} purchases still need a live read")

    total = len(still_needed) + len(p_rows)
    if total:
        with sync_playwright() as p:
            ctx = browser.open_reader_context(p, "motorway", headless=True)
            try:
                page = ctx.pages[0] if ctx.pages else ctx.new_page()
                for i, r in enumerate(still_needed):
                    try:
                        d = motorway.read_vehicle_details(page, r["listing_url"])
                    except Exception as e:
                        print(f"  [{i+1}/{total}] {r['reg']}: FAILED, {e}")
                        continue
                    if r.get("mileage") is None and d["mileage"] is not None:
                        db.set_lost_bid_mileage(r["id"], d["mileage"])
                    db.set_lost_bid_fields(r["id"], transmission=d["transmission"], engine=d["engine"])
                    done += 1
                    if (i + 1) % 25 == 0 or i + 1 == len(still_needed):
                        print(f"  [{i+1}/{total}] {r['reg']}: {d['mileage']} miles, "
                              f"{d['transmission']}, {d['engine']}")
                for j, r in enumerate(p_rows):
                    i = len(still_needed) + j
                    try:
                        d = motorway.read_vehicle_details(page, r["listing_url"])
                    except Exception as e:
                        print(f"  [{i+1}/{total}] {r['reg']}: FAILED, {e}")
                        continue
                    db.set_purchase_fields(r["id"], transmission=d["transmission"], engine=d["engine"])
                    done += 1
                    if (j + 1) % 25 == 0 or j + 1 == len(p_rows):
                        print(f"  [{i+1}/{total}] {r['reg']}: {d['transmission']}, {d['engine']} (purchase)")
            finally:
                ctx.close()
    print(f"Details: {done} read live, {from_db} filled from a sighting already on file")


def run_retail(min_samples, limit):
    keys = _target_keys(min_samples)
    rows = [r for r in db.lost_bids_missing_retail()
            if bid_suggest.model_key(r["name"]) in keys]
    if limit:
        rows = rows[:limit]
    print(f"{len(rows)} lost bids have a mileage but need a Cazana retail lookup")
    if not rows:
        return
    done = 0
    with sync_playwright() as p:
        for i, r in enumerate(rows):
            parts = (r.get("name") or "").split(None, 1)
            make = parts[0] if parts else ""
            model = parts[1] if len(parts) > 1 else make
            try:
                val, _url = cazana.lookup(p, r["reg"], r["mileage"], expect=(make, model))
            except Exception as e:
                print(f"  [{i+1}/{len(rows)}] {r['reg']}: FAILED, {e}")
                continue
            db.set_lost_bid_retail(r["id"], val)
            done += 1
            print(f"  [{i+1}/{len(rows)}] {r['reg']}: {'£' + str(val) if val is not None else 'no match'}")
    print(f"Retail: {done} of {len(rows)} updated")


def main():
    min_samples = 10
    if "--min-samples" in sys.argv:
        i = sys.argv.index("--min-samples")
        min_samples = int(sys.argv[i + 1])
    limit = None
    if "--limit" in sys.argv:
        i = sys.argv.index("--limit")
        limit = int(sys.argv[i + 1])

    if "--fields" in sys.argv:
        run_fields(min_samples, limit)
    elif "--mileage" in sys.argv:
        run_mileage(min_samples, limit)
    elif "--retail" in sys.argv:
        run_retail(min_samples, limit)
    else:
        print("Usage: python3 bid_retail_enrich.py --fields|--mileage|--retail [--limit N] [--min-samples N]")
        sys.exit(1)


if __name__ == "__main__":
    main()
