#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
shape_compression.py — can the SHAPE of a combination shrink the space for free?
================================================================================
A fair draw (see the fairness audit) is a uniform sample of all combinations, so
every shape region should hold the same fraction of *winners* as of *combos* —
i.e. you can't cut space faster than you cut history. This script measures that,
for Joker (5/45) and Eurojackpot (5/50):

  1. per-feature "safe-trim lift" = history kept / space kept at 99% recall, for
     genuine shape features (functions of the 5 numbers) — expected ~1.0;
  2. the frame-mismatch trap: point-in-time features (E*/ED*) whose history and
     space ranges don't even overlap, giving a spurious "huge cut";
  3. the multi-dimensional mirage: as you add feature-dimensions, more of the
     space looks "empty" of history — but only because cells go unsampled
     (curse of dimensionality), never because draws avoid them.

Needs each game's tot_df_dynamic_basic.parquet (space) + hist_df.csv (history).
Writes 2 PNGs to static/img/blog/shape-compression/ and prints the summary.
"""
from __future__ import annotations
from pathlib import Path
from collections import Counter

import numpy as np
import pandas as pd
import matplotlib
matplotlib.use("Agg")
import matplotlib.pyplot as plt

GAMES = {"Joker (5/45)": "games/joker/data", "Eurojackpot (5/50)": "games/eurojackpot/data"}
COL = {"Joker (5/45)": "#e0a417", "Eurojackpot (5/50)": "#3ba7d6"}
CLEAN = ["sum", "range", "AC", "SD", "cluster_manhattan_mean", "cluster_euclid_mean",
         "bbox_area", "bbox_perim", "Cols", "sameDs", "D<=10", "D<=20", "Syn", "stidx"]
MISMATCH = ["E0", "E1", "ED1"]
DIMS = ["sum", "range", "cluster_euclid_mean", "AC"]
IMG = Path("static/img/blog/shape-compression"); IMG.mkdir(parents=True, exist_ok=True)
plt.rcParams.update({"figure.dpi": 130, "font.size": 11, "axes.grid": True, "grid.alpha": .25})


def load(d):
    tot = pd.read_parquet(f"{d}/tot_df_dynamic_basic.parquet"); tot.columns = [str(c).strip() for c in tot.columns]
    tot = tot.loc[:, ~tot.columns.duplicated()]
    hist = pd.read_csv(f"{d}/hist_df.csv"); hist.columns = [str(c).strip() for c in hist.columns]
    hist = hist.loc[:, ~hist.columns.duplicated()]
    return tot, hist


def lift(h, s):
    lo, hi = np.percentile(h, [0.5, 99.5])
    sk = ((s >= lo) & (s <= hi)).mean()
    return (0.99 / sk if sk > 0 else np.nan), 1 - sk


def multidim_empty(tot, hist, cols, bins):
    S = tot[cols].to_numpy(float); H = hist[cols].to_numpy(float); nS, nH = len(S), len(H)
    edges = [np.quantile(S[:, i], np.linspace(0, 1, bins + 1)) for i in range(len(cols))]
    si = tuple(np.clip(np.digitize(S[:, i], edges[i][1:-1]), 0, bins - 1) for i in range(len(cols)))
    hi = tuple(np.clip(np.digitize(H[:, i], edges[i][1:-1]), 0, bins - 1) for i in range(len(cols)))
    sc = Counter(zip(*si)); hc = Counter(zip(*hi))
    empty_sf = 0.0; hi_empty_sf = 0.0
    for cell, cnt in sc.items():
        sf = cnt / nS
        if hc.get(cell, 0) == 0:
            empty_sf += sf
            if sf * nH >= 5:
                hi_empty_sf += sf
    return empty_sf, hi_empty_sf


d1fig, d1ax = plt.subplots(1, 2, figsize=(11, 4.3))
d2fig, d2ax = plt.subplots(1, 2, figsize=(11, 4.3))

for gi, (game, d) in enumerate(GAMES.items()):
    tot, hist = load(d); c = COL[game]
    print(f"\n=== {game} ===")
    print(" genuine shape features — safe-trim lift (history kept / space kept, 99% recall):")
    lifts = []
    for f in CLEAN:
        if f in tot.columns and f in hist.columns:
            lf, cut = lift(pd.to_numeric(hist[f], errors="coerce").dropna().to_numpy(float),
                           pd.to_numeric(tot[f], errors="coerce").dropna().to_numpy(float))
            lifts.append(lf)
            print(f"   {f:<24} lift={lf:.3f}  (space cut {100*cut:4.1f}%)")
    print(f"   -> clean-shape lift: mean={np.nanmean(lifts):.3f} median={np.nanmedian(lifts):.3f} max={np.nanmax(lifts):.3f}")
    print(" frame-mismatch trap (point-in-time 'features'):")
    for f in MISMATCH:
        if f in tot.columns and f in hist.columns:
            hr = (float(hist[f].min()), float(hist[f].max())); sr = (float(tot[f].min()), float(tot[f].max()))
            print(f"   {f:<6} history range {hr}  vs  space range {sr}  -> incompatible, lift is spurious")

    # img1: sum — history vs space distribution (they coincide -> nothing to trim)
    ax = d1ax[gi]
    s = pd.to_numeric(tot["sum"], errors="coerce").dropna(); h = pd.to_numeric(hist["sum"], errors="coerce").dropna()
    bins = np.linspace(min(s.min(), h.min()), max(s.max(), h.max()), 45)
    ax.hist(s, bins=bins, density=True, color="#999", alpha=.6, label="all combinations (space)")
    ax.hist(h, bins=bins, density=True, histtype="step", color=c, lw=2, label="real draws (history)")
    ax.set_title(f"{game}\nsum: winners fill the space evenly", fontsize=10)
    ax.set_xlabel("sum of the 5 numbers"); ax.set_ylabel("density"); ax.legend(fontsize=9)

    # img2: multi-dimensional emptiness mirage
    empty = []; hiempty = []
    for k in (1, 2, 3, 4):
        bins = {1: 40, 2: 16, 3: 10, 4: 7}[k]
        e, he = multidim_empty(tot, hist, DIMS[:k], bins)
        empty.append(100 * e); hiempty.append(100 * he)
    ax = d2ax[gi]
    ax.plot([1, 2, 3, 4], empty, "o-", color=c, lw=2, label="space in EMPTY cells")
    ax.plot([1, 2, 3, 4], hiempty, "s--", color="k", lw=1.6, label="…that should have had draws (exp≥5)")
    ax.set_xticks([1, 2, 3, 4]); ax.set_ylim(bottom=-2)
    ax.set_title(f"{game}\nthe multi-dimensional mirage", fontsize=10)
    ax.set_xlabel("feature dimensions"); ax.set_ylabel("% of space"); ax.legend(fontsize=9)
    print(f" multi-dim empty space%: {[round(x,1) for x in empty]}  |  should-have-draws empty%: {[round(x,2) for x in hiempty]}")

for fig, name in [(d1fig, "img1"), (d2fig, "img2")]:
    fig.tight_layout(); fig.savefig(IMG / f"{name}.png", bbox_inches="tight"); plt.close(fig)
print(f"\ncharts -> {IMG}")
