"""
CompMatch: BidBrain's own comp based suggested bid engine (named 2026-08-25,
Mark: "give the ai suggested bid a name so i can come back to improve
later"). Learned from BidBrain's own real history (Mark 2026-08-24: "the
original vision for BB was to use ai to understand from our own data our
best sellers... to make educated bids", then, once the Didn't win data
existed, "lets try and build a AI suggested bid to add to the slider that
should win the car based on both won and lost (but sold for) data").
Refined 2026-08-25, same conversation: "compare year, transmission, engine
size, mileage so that the ai is comparing as similar cars as it can".

Deliberately simple and explainable rather than a real trained model: a
suggestion is the 70th percentile of real winning prices among the closest
real comps for that make and model, ranked by how similar their own
mileage, year, transmission and engine size are to the car being priced,
using only whichever of those a comp and the target car both happen to
have (never a guess, never a penalty for a historical gap). A won
purchase's own winning_bid is exactly as real a data point as a lost
auction's own sold_for, both are the true price that won that specific
car; db.bid_comps_full() already combines them with whatever else is
known about each one.

Never a gate, never touches pricing.py's own formula. A suggestion is
shown only when there is real comparable history (min_samples), otherwise
none, the same "hold back rather than guess" spirit as the rest of this
project.
"""

import re
import statistics
from datetime import date

_ENGINE_LITRES_RE = re.compile(r"(\d\.\d)")

# A comp older than this never contributes a price figure at all (Mark
# 2026-08-25: "ok have another go at a model for suggested bid").
# Investigated live: the Didn't win history reaches back 58 MONTHS, and
# because every comp's Cazana retail was looked up NOW while its sale
# happened up to 5 years ago, depreciation systematically inflates old
# comps' own %-of-retail figures, measured month by month against the
# real data: median ratio 0.76 for bids under a month old, climbing
# steadily to 0.81 around 18 months, 0.92 by two years, and 1.2+ at
# three to five years (past 40 months, 100% of comps exceeded the
# MAX_PLAUSIBLE_RATIO cap). So the "stale or mismatched Cazana lookups"
# first blamed for implausible ratios are mostly just OLD comps priced
# against today's depreciated retail. Old raw amounts are even more
# wrong, the market and the car's own age have both moved on, and they
# previously flowed into the flat fallback completely unfiltered. 18
# months keeps the measured drift small (about +0.05 on the median
# ratio, versus +0.45 unfiltered) while live coverage actually ROSE
# versus the pre-window model (126 to 130 real cars on the 2026-08-25
# list, the naming and gearbox-bucket fixes below more than paying for
# the comps the window removes).
COMP_MAX_AGE_MONTHS = 18

_DAYS_PER_MONTH = 30.44

# A comp whose own winning price came to more than this share of its own
# retail is excluded from every % of retail calculation (Mark 2026-08-25,
# checked live: about a third of real comps for some models show a
# winning price above their own retail, almost certainly stale or
# mismatched historical Cazana lookups rather than genuine overpaying,
# and badly distort a same-model comparison if left in, see
# suggest_bid_for_car's own docstring for the before/after check). 1.05
# allows a little real headroom (a genuinely hot bidding environment can
# nudge slightly over retail) without trusting anything further out.
MAX_PLAUSIBLE_RATIO = 1.05


def model_key(name: str) -> str:
    """A coarse "Make Model" grouping key from a free text vehicle name
    (for example "Mercedes A 180 D AMG Line Premium Auto" and "Mercedes A
    180 D Sport Premium Auto" both key to "MERCEDES A"), the first two
    words, upper cased. Checked live against the real combined history
    2026-08-24: 55 of 113 real distinct keys already carry 5 or more real
    comps, 31 carry 10 or more, without needing anything finer. A one word
    name (should not happen in practice, defensive only) keys on itself.

    Canonicalised 2026-08-25, the same real problem render.py's own
    _model_family/_canon_make_map already solved for the cockpit's Make
    and Model filters, found live to be splitting real comp groups so
    badly that whole models never got a suggestion at all: BMW comp
    names carry the engine badge as the model word ("118D", "320D"), so
    every BMW comp keyed to its own tiny badge group while a live "BMW 1
    Series" car keyed to "BMW 1" and found ZERO of the 27 real 1 Series
    comps on file; Mercedes was split three ways ("MERCEDES A" 34,
    "MERCEDES A180" 7, "MERCEDES-BENZ A-CLASS" 8, one real model, 51
    comps once merged); a live "MG Motor UK ZS" keyed to "MG MOTOR" and
    missed the real "MG ZS" group; and a handful of comps carry
    Motorway's own truncated "Insig" for Insignia. Same regex shapes as
    render._model_family (deliberately duplicated, not imported, this
    module stays pure and render depends the other way), applied to the
    one combined name string both comps and live cars key through, so
    both sides always land on the same canonical group."""
    parts = (name or "").upper().split()
    if not parts:
        return ""
    make, rest = parts[0], parts[1:]
    if make == "MERCEDES-BENZ":
        make = "MERCEDES"
    if make == "MERCEDES" and rest and rest[0] == "BENZ":
        # A make and model pair joined into one name spells the make with a
        # space ("Mercedes Benz" "A Class") where a listing title spells it
        # with a hyphen ("MERCEDES-BENZ A-CLASS"). One real make either way,
        # so both land on the same key (2026-09-20, added when the stock
        # badge and the top picks started keying through this function).
        rest = rest[1:]
        if not rest:
            return make
    if make == "MG" and rest and rest[0] == "MOTOR":
        # "MG Motor UK ZS" is the make "MG Motor UK" plus the model "ZS".
        rest = rest[1:]
        if rest and rest[0] == "UK":
            rest = rest[1:]
    if not rest:
        return make
    model = rest[0]
    if make == "BMW":
        # "118D" / "320D" / "116I" are the 1/3 Series engine badges; a
        # live car's own model field says "1 Series" so both collapse to
        # the series digit. X1/X2... deliberately untouched.
        m = re.match(r"^(\d)\d{2}[A-Z]{0,2}$", model)
        if m:
            model = m.group(1)
    elif make == "MERCEDES":
        # "A-CLASS" (a live car's own model field) and "A180"/"CLA220"
        # (a comp name's trim badge) are both the class itself.
        m = re.match(r"^([A-Z]{1,3})-?CLASS$", model) or re.match(r"^([A-Z]{1,3})\d{3}$", model)
        if m:
            model = m.group(1)
    elif make == "VAUXHALL" and model == "INSIG":
        model = "INSIGNIA"  # Motorway's own truncated Insignia, seen live on 2 comps
    return f"{make} {model}"


def engine_litres(text: str):
    """The engine size in litres parsed out of a free text engine string
    ("2.0 TDI", "1.6 Diesel Automatic"), or None when no plain X.X figure
    is found. Never guessed from anything else (a badge like "320d" is not
    parsed as 3.2)."""
    m = _ENGINE_LITRES_RE.search(text or "")
    return float(m.group(1)) if m else None


_REG_AGE_RE = re.compile(r"^[A-Z]{2}(\d{2})[A-Z]{3}$")


def year_from_reg(reg: str):
    """A real registration year straight off a current format UK plate's
    own age identifier (Mark 2026-08-25, "make sure the age and mileage
    are being considered", checked live: sightings only covers 5 of 1038
    real comps, far too sparse, while 980 of 1038 real regs already match
    this format). Format is AA99AAA: the two digits are 01-50 for a plate
    issued March to August of 2000+that figure, or 51-99 for one issued
    September to February of 2000+(that figure minus 50). Whole year
    only, not the exact month, which is all the comparison model needs.
    None for any other plate shape (an older letter prefix or suffix
    style, a private plate, or an invalid 00 identifier), never guessed."""
    reg = (reg or "").upper().replace(" ", "")
    m = _REG_AGE_RE.match(reg)
    if not m:
        return None
    d = int(m.group(1))
    if 1 <= d <= 50:
        return 2000 + d
    if 51 <= d <= 99:
        return 2000 + (d - 50)
    return None


_AUTO_NAME_RE = re.compile(
    r"\b(Automatic|Auto|DSG|Tiptronic|S[\s-]?Tronic|PDK|CVT|S-A)\b", re.I)


def transmission_from_name(name: str):
    """"Automatic" when the platform's own free text vehicle name carries
    a real automatic/semi-automatic badge (Auto, DSG, Tiptronic, S Tronic,
    PDK, CVT, or Motorway's own "S-A" abbreviation for Semi-Automatic,
    checked live: 120 of 1038 real comp names carry one of these), the
    same "Automatic" or "Semi auto" or nothing pricing.py's own
    _gearbox_label already uses for a live car, collapsed to one bucket
    here since a comp's exact semi vs full automatic split is not needed
    for the similarity match, only automatic vs manual is. Never returns
    "Manual": these names never spell that out (0 of 1038 do), so an
    unlabelled car is left unknown rather than assumed manual, the same
    "never guess" rule as everywhere else in this module."""
    return "Automatic" if _AUTO_NAME_RE.search(name or "") else None


def comp_age_months(date_text, today=None):
    """How many months ago a comp's own sale happened, from its date_bid
    (lost bids) or bought_date (purchases), both plain YYYY-MM-DD strings.
    None when the date is missing or unreadable, never guessed. today is
    overridable for tests only, the same convention as pricing.assess."""
    try:
        d = date.fromisoformat((date_text or "")[:10])
    except ValueError:
        return None
    return ((today or date.today()) - d).days / _DAYS_PER_MONTH


def _tx_bucket(t):
    """Manual vs automatic, the two buckets the hard transmission gate
    actually compares on. Semi-automatic buckets WITH automatic: it is an
    automatic from a buyer's (and a value) point of view, and pricing.py's
    own class rules already treat "Automatic or Semi automatic" as one
    family (the no automatic Fords rule). Manual stays pure, nothing else
    ever buckets into it, so Mark's own "never mix manual and auto" hard
    gate is fully respected, this only stops the 13 real Semi-automatic
    comps (and every live Semi-automatic car, 24 on the 2026-08-25 list)
    being stranded in a third bucket too small to ever match anything.
    Checked live: the only real values anywhere in the data are Manual,
    Automatic, Semi-automatic and None, but matched on the word rather
    than exact equality so a future "6 speed manual" style string could
    never silently bucket as automatic. None (unknown) stays None."""
    t = str(t or "").strip().lower()
    if not t:
        return None
    return "manual" if "manual" in t else "automatic"


def _percentile(sorted_values, pct):
    """Linear interpolated percentile of an already sorted list, pct 0-100.
    Plain and deterministic rather than statistics.quantiles, which
    changes its own boundary behaviour across small sample sizes in a way
    that is harder to reason about for a business figure like this."""
    if not sorted_values:
        return None
    if len(sorted_values) == 1:
        return sorted_values[0]
    k = (len(sorted_values) - 1) * (pct / 100.0)
    lo, hi = int(k), min(int(k) + 1, len(sorted_values) - 1)
    frac = k - lo
    return sorted_values[lo] + (sorted_values[hi] - sorted_values[lo]) * frac


def build_pool(comps_full) -> dict:
    """Groups db.bid_comps_full()'s own rows (name, amount, retail,
    mileage, year, transmission, engine, any but name and amount may be
    None) by model_key. Returns {model_key: [comp dict, ...]}."""
    pool = {}
    for c in comps_full:
        if c.get("amount") is None or not c.get("name"):
            continue
        pool.setdefault(model_key(c["name"]), []).append(c)
    return pool


def _similarity(target: dict, comp: dict) -> float:
    """A similarity score between the car being priced and one real comp,
    both plain dicts with mileage/year/transmission/engine/grade. Only
    ever scores a factor when BOTH sides have it, so a gap in the
    historical record never counts against (or for) a comp, it just
    carries less weight. Mileage is weighted heaviest, it is the single
    biggest real driver of what a specific car is worth within a model;
    transmission is a binary match (Mark 2026-08-25: "if its manual or
    auto as auto will be worth more"); engine size, year and condition
    grade all taper off gently rather than a hard cutoff, since a comp a
    year, half a litre or one grade off is still a useful data point,
    just a slightly less exact one."""
    score = 0.0
    if target.get("mileage") and comp.get("mileage"):
        score += max(0.0, 2.0 - abs(target["mileage"] - comp["mileage"]) / 15000.0)
    if target.get("year") and comp.get("year"):
        score += max(0.0, 1.0 - abs(target["year"] - comp["year"]) / 4.0)
    if target.get("transmission") and comp.get("transmission"):
        if str(target["transmission"]).strip().lower() == str(comp["transmission"]).strip().lower():
            score += 1.0
    t_lit, c_lit = engine_litres(target.get("engine")), engine_litres(comp.get("engine"))
    if t_lit and c_lit:
        score += max(0.0, 1.0 - abs(t_lit - c_lit) / 0.6)
    if target.get("grade") and comp.get("grade"):
        score += max(0.0, 1.0 - abs(target["grade"] - comp["grade"]) * 0.5)
    return score


def suggest_bid_for_car(car, pool: dict, min_samples: int = 5, top_n: int = 15,
                         car_retail=None, min_ratio_samples: int = 3, today=None):
    """The suggested bid for this exact car (needs make, model, and
    ideally mileage/year/transmission/engine/grade, any of which may be
    blank), ranked against real comps for its make and model by how
    closely each one's own mileage, year, transmission, engine size and
    condition grade matches, not just "same model" (Mark 2026-08-25: "as
    similar as it can").

    Narrows to the closest top_n comps only once there is real similarity
    signal to rank on (at least min_samples comps that share at least one
    comparable field with this car), otherwise falls back to every real
    comp for the model, the same plain "what did this model actually go
    for" figure the original build used, so an early, still sparsely
    enriched history never gets narrowed down to fewer than the reliable
    floor.

    Within that same closest cohort, when car_retail (this car's own real
    Cazana retail value) is known AND at least min_ratio_samples of the
    cohort also carry a real retail figure of their own, the suggestion
    is computed as a % of retail rather than a flat price (Mark
    2026-08-25: "if we also knew that a 3 series winning bid was 90% of
    cazana retail... having this info should improve the accuracy"): the
    70th percentile of each comp's own winning-price-over-its-own-retail
    ratio, applied to THIS car's own retail value, so two comps of very
    different spec (and so very different retail) are compared on how
    competitively each one actually sold relative to its own worth,
    rather than treated as equally comparable raw prices. Falls back to
    the plain raw-price percentile whenever this car's own retail is not
    yet known (no Cazana match) or too few comps in the cohort carry one.

    Whatever the source, the final figure is never allowed to exceed this
    car's own retail value when it is known (Mark 2026-08-25, a real
    car found live: "the ai bids are way off atm... manual is worth
    less than automatic, more mileage = low value, new the car means
    its worth more". Investigated: for the model in question 14 of 25
    real comps showed a winning price ABOVE their own retail, most
    likely stale or mismatched historical Cazana lookups rather than
    dealers genuinely overpaying at that scale, but regardless of the
    cause, a suggestion to pay more than a car is actually worth can
    never be sound buying advice, so it is capped, the same "play safe"
    discipline as the rest of this project). The same implausible-ratio
    comps (over MAX_PLAUSIBLE_RATIO) are also excluded from every % of
    retail calculation in the first place, not just clamped at the end:
    checked live, they badly distort a same-model comparison, before
    excluding them 3 of 4 real models with enough automatic-vs-other
    data showed automatics selling for LESS of their own retail than
    everything else, backwards from reality; after excluding them, 3 of
    4 correctly showed automatics selling for more, matching Mark's own
    "auto will be worth more".

    Manual and automatic are NEVER mixed together, a hard gate, not a
    soft preference (Mark 2026-08-25, first "really considering what %
    of cazana retail each make model and transmission each car
    achieves", then, once the first version of this still sometimes
    fell back to a mixed pool when there was not quite enough of one
    transmission: "the suggested bid must hard gate respect the
    difference between manual and auto, it would be better to have less
    data than inc auto and manual together"). Whenever this car's own
    transmission is known (virtually always true for a live car read
    straight off a platform), candidates are narrowed to ONLY comps that
    ALSO have a known, matching transmission, before anything else runs,
    with NO fallback to a mixed pool, ever, even if that leaves too
    little data to suggest anything at all, in which case this returns
    None rather than guess from a diluted comparison. Only when this
    car's own transmission is genuinely unknown (should be rare) does
    the gate not apply, there is nothing to respect the difference
    against. Deliberately never a CROSS-model adjustment either (tried,
    checked live, rejected: it mixed in whichever models happen to get
    badged "Auto" in their own listing name, a real confound, different
    models, not the same model on two gearboxes, and gave a backwards,
    wrong signal), always compared within one model at a time.

    The transmission match is on the manual vs automatic BUCKET
    (_tx_bucket), not the exact platform string: Semi-automatic is an
    automatic for value purposes and matches the automatic comps, manual
    stays pure either way, so the hard gate above is fully intact.

    A comp older than COMP_MAX_AGE_MONTHS, or with no readable sale date
    at all, never contributes a price figure, ratio or raw (see the
    constant's own comment for the measured month by month drift that
    forced this: today's Cazana retail against a years old sale price
    systematically inflates an old comp's own ratio, and an old raw
    amount is a price from a market and a car age that no longer exist).
    An undated comp is excluded rather than assumed recent, hold back
    rather than guess. Within the window, fresher comps also rank
    slightly ahead of equally specced older ones.

    Returns {"n", "median", "suggested", "like_for_like", "ratio_based",
    "capped", "transmission_matched"}, or None when there is not even
    min_samples worth of recent enough comps left for this model (and
    this car's own transmission, once the hard gate applies) at all,
    never a guess off a handful of cars."""
    key = model_key(f"{getattr(car, 'make', '') or ''} {getattr(car, 'model', '') or ''}")
    candidates = pool.get(key, [])
    target = {
        "mileage": getattr(car, "mileage", None),
        "year": getattr(car, "year", None),
        "transmission": getattr(car, "transmission", None),
        "engine": getattr(car, "engine", None),
        "grade": getattr(car, "grade", None),
    }

    transmission_matched = False
    if target.get("transmission"):
        want = _tx_bucket(target["transmission"])
        candidates = [c for c in candidates if _tx_bucket(c.get("transmission")) == want]
        transmission_matched = True

    ages = {id(c): comp_age_months(c.get("date"), today) for c in candidates}
    candidates = [c for c in candidates
                  if ages[id(c)] is not None and ages[id(c)] <= COMP_MAX_AGE_MONTHS]

    if len(candidates) < min_samples:
        return None

    # Rank by spec similarity plus a small recency bonus (0.5 for a sale
    # this month tapering to 0 at the window edge, deliberately smaller
    # than any single spec factor so it breaks ties toward the current
    # market rather than outvoting a genuinely closer matched car). The
    # like_for_like decision stays on spec similarity ALONE, recency is
    # not a spec match and every windowed comp has a date, so counting it
    # would wrongly flag every suggestion as like for like.
    def _recency(c):
        return max(0.0, 0.5 - ages[id(c)] / (2 * COMP_MAX_AGE_MONTHS))
    scored = [(_similarity(target, c), c) for c in candidates]
    ranked_n = sum(1 for s, _ in scored if s > 0)
    if ranked_n >= min_samples:
        scored.sort(key=lambda pair: -(pair[0] + _recency(pair[1])))
        use = [c for _, c in scored[:max(top_n, min_samples)]]
        like_for_like = True
    else:
        use = candidates
        like_for_like = transmission_matched

    # The comps actually used, in ranking order, for the cockpit's detail
    # drawer (Steven 2026-09-02, Step 5 of the review: "CompMatch's
    # reasoning with the real comparable sales it used"). Only the fields
    # a person reads off a comparable, the same eight the comp pool holds,
    # never a registration. A plain list of small dicts so it serialises
    # straight into the card's own JSON.
    comps_used = [{"name": c.get("name"), "amount": c.get("amount"), "retail": c.get("retail"),
                   "mileage": c.get("mileage"), "year": c.get("year"),
                   "transmission": c.get("transmission"), "date": c.get("date")}
                  for c in use[:8]]
    ratios = sorted(c["amount"] / c["retail"] for c in use
                     if c.get("retail") and c.get("amount") is not None
                     and c["amount"] / c["retail"] <= MAX_PLAUSIBLE_RATIO)
    if car_retail and len(ratios) >= min_ratio_samples:
        suggested = round(car_retail * _percentile(ratios, 70))
        return {
            "n": len(ratios),
            "median": statistics.median(ratios) * car_retail,
            "suggested": min(suggested, car_retail),
            "like_for_like": like_for_like,
            "ratio_based": True,
            "capped": suggested > car_retail,
            "transmission_matched": transmission_matched,
            "comps": comps_used,
        }
    amounts = sorted(float(c["amount"]) for c in use)
    suggested = round(_percentile(amounts, 70))
    return {
        "n": len(use),
        "median": statistics.median(amounts),
        "suggested": min(suggested, car_retail) if car_retail else suggested,
        "like_for_like": like_for_like,
        "ratio_based": False,
        "capped": bool(car_retail) and suggested > car_retail,
        "transmission_matched": transmission_matched,
        "comps": comps_used,
    }
