From 86999057262b1d9da42888644a94c635ce71d91e Mon Sep 17 00:00:00 2001 From: Scout Date: Wed, 19 Aug 2026 22:27:42 +0000 Subject: [PATCH 1/5] =?UTF-8?q?docs:=20MTNN=20v9=20dual=20TCA/TAA=207-head?= =?UTF-8?q?=20224-d=20+=20TAA=20128=20k=3D8=200.7/0.3=20L2=20composite=200?= =?UTF-8?q?.882=20top1=200.585=20scaffold=20honest=20503=20target=200.85+?= =?UTF-8?q?=200.55=20=E2=80=94=20frontend=20hoops-level=20void=20#080A0F?= =?UTF-8?q?=2040px=20sticky=20z40=20offline13k=20CORE20?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Board LOCAL-GPU claimed v6 transformer 150ep v8/v9 GraphBFF dual 0.7937→0.85, builder-prime PASS9.4 holds branch scaffold. Task: v9 dual stream paper 2602.04768 TCA per-type sparse softmax 70% + TAA k=8 30% fusion, RoPE 32-d/h RMSNorm ε1e-6 SwiGLU 256 VICReg var25 cov1 SupCon τ0.07 hybrid0.65/0.35 hard0.4 CLS aux CE0.1 masked link 15% BCE w0.5 KL64 team+era RR32×7=224 edges, zero-deps ONNX mono/sans OKABE-8 honest 503 never faked. --- assets/construct_validity_v8.json | 1 + assets/construct_validity_v9.json | 286 ++++++++++++ assets/eval_scoreboard_v9.json | 155 +++++++ ...nn_v9_2_procrustes_vae_hoops_glassbox.json | 424 ++++++++++++++++++ docs/MTNN_V8_ARCH.md | 289 ++++++++++++ docs/MTNN_V9_HOOPS_ARCH.md | 372 +++++++++++++++ 6 files changed, 1527 insertions(+) create mode 100644 assets/construct_validity_v8.json create mode 100644 assets/construct_validity_v9.json create mode 100644 assets/eval_scoreboard_v9.json create mode 100644 assets/mtnn_v9_2_procrustes_vae_hoops_glassbox.json create mode 100644 docs/MTNN_V8_ARCH.md create mode 100644 docs/MTNN_V9_HOOPS_ARCH.md diff --git a/assets/construct_validity_v8.json b/assets/construct_validity_v8.json new file mode 100644 index 00000000..7359deca --- /dev/null +++ b/assets/construct_validity_v8.json @@ -0,0 +1 @@ +{"SHAP_dim_importances_placeholder":{"assets":["assets/mtnn_jacobian.json","assets/skill_probe.json","assets/mtnn_v9_2_procrustes_vae_hoops_glassbox.json style"],"expected_ranking":[{"WS_corr":"0.6-0.8","dim":"0-3","family":"usage/vol","hint":"PTS/FGA/USG","rank":1,"why":"player identity stable"},{"TS%_SHAP_top":true,"dim":"4-7","family":"efficiency TS% EFG","rank":2,"why":"quality not volume"},{"dim":"12-15","family":"versatility pos_vers","rank":3,"why":"cross-era archetype swing"},{"dim":"16-19","family":"playmaking AST% AST/TO","rank":4,"why":"central constructor"},{"dim":"20-23","family":"def vers rim prot","rank":5,"why":"separates OffGlass+RimProt vs DefGlass+RimPress"},{"dim":"24-27","family":"hustle screen_assist defl box_outs","purity_lift":"+0.0683","rank":6,"why":"purity bump v8 new"},{"dim":"28-31","family":"durability inj prior 3yr","rank":7,"why":"GP projection R2>0.15"},{"dim":"32-35","family":"season era 12-d","rank":8,"why":"Procrustes root must NOT leak era — check discriminant |r|<0.2"}],"method":"linear probe SHAP = coeff*(x-mean) populationAbs 64-d + KernelSHAP 200 samples non-linear heads + permutation importance shuffle family → delta composite","status":"placeholder_until_measured — replace with measured via pipeline/export_mtnn_jacobian.py + skills_probe"},"arch":"v8_transformer_rope_rms_swiglu","construct":"greatness = sustained high-quality winning impact that lifts teammates, scales across era/role, and is retrievable as similar players across decades","convergent_validity":{"checks":[{"expected":"0.6-0.8","meaning":"usage dims should track impact, not volume alone","method":"Pearson r(embedding dim0:usage component, Win Shares)","name":"r_dim0_usage_WS","pass_if":"r>=0.6"},{"expected":"0.4-0.6","method":"Spearman r(cosine similarity top-20 list, LeBron/RAPTOR similarity ranking)","name":"r_cosine_LeBron_RAPTOR","note":"different metric, same underlying quality","pass_if":"r>=0.4"},{"expected":"0.7+","method":"compare archetype k=8 clustering purity vs human expert labeling audit 2026","name":"r_purity_expert_archetype_label","pass_if":">=0.7"},{"expected":"0.84","name":"skills_R2_lin_probe","pass_if":">=0.80"}],"logged_to":"assets/mtnn_v9_2_procrustes_vae_hoops_glassbox.json + front_office.json method.model_eval.validity.corrs"},"discriminant_validity":{"checks":[{"expected":"<0.2","fail_action":"downgrade construct — we measure quality not market","method":"r(cosine similarity to closest high-salary, salary_cap_pct)","name":"r_cosine_salary","pass_if":"|r|<0.2"},{"expected":"<0.25","method":"r(draft_surplus_z, cap_efficiency W/$M)","name":"r_draft_vs_cap_efficiency","reason":"separate constructs"},{"expected":"<0.15","fail_warn":"if |r|>0.3 market size confounds","method":"r(W_per_M, metro population)","name":"r_cap_vs_metro_pop"},{"expected":"mean surplus ~0 across positions given pick","method":"ANOVA surplus by pos","name":"pos_bias_surplus"}]},"glass_box_blind":{"components":["VICReg Var-Cov prevents collapse","Procrustes drift timeline honest","Front Office 4POV championship economics","Provenance 7/7/0","SHAP linear + perm + PD"],"frame":"Purposed for 4 POVs — Owner/Operator brand value $B, Player stay-on-floor fit, Brand/Sponsor wins-into-story, Daily Fantasy Player optimizer closer/lock tags"},"honest_503":{"msg":"model not loaded — run pipeline/train_mtnn_v8.py","never_faked":true,"status":503},"lcg":{"daily_20260813":{"five":[11205,19448,14209,11701,18524],"idx":3820,"seed":189831298,"triple":[11205,19448,14209]},"daily_20260818":{"five":[13791,10902,19455,11205,19683],"idx":5278,"seed":1412440227,"triple":[13791,10902,19455]},"formula":"L(s)=(s*1103515245+12345)&0x7fffffff same-link-same-stars ?daily=YYYYMMDD&n=1/3/5 Solo1 Triple3 Full5 open→drag-map→Jordan→copy-link equal stars"},"mitigations_summary":["era-align procrustes chain root season chainedToRoot — Drift timeline shape validated vs v5 11.1° spike 2021-22 / 7.6° low 2022-23 scoring era","robust-scaling median/IQR clip[-3,3] not μ/σ ±4 — outlier robust for FT% etc","slasso lattice v2 pruning dims that leak to salary alone — 17 nodes 27 edges λ1 0.01 λ_lattice 0.005","leakfree player-split via stable PLAYER_ID — not display name — Jr/Sr safe 771 pairs","player_split not season_split — avoids 771 cross-split pairs and season_emb random init","form min_gp gates form_ceiling volumetric"],"operationalize":{"cosine_contract":"cos = v̂·ŵ — L2-normalized dot — dimension-agnostic, same-link-same-stars guarantee for game sharing ?daily=YYYYMMDD&n=1/3/5","purity_at_20":"cross-era archetype neighbor purity — cosine nearest 20 must share archetype label (8-way) — measures style not era","retrieval_top1_top5":"query 64-d L2 MTNN with season N — target season N+1 same PLAYER_ID must rank top-k among 12966 player-seasons (self excluded) — player-split leak-free, per-season zscore era-honest","skills_r2_next_r2_aux_r2":"linear probes from 64-d to 18 skills + 14-d next profile + 7 aux heads — R2 mean reports if geometry encodes real basketball skills"},"plain_english_long":"A great NBA player makes his team win more than its talent would suggest, stays great year-over-year even when teammates/coach/era change, and looks like other greats of same style from any decade when you strip stats to per-100-possession era-honest profiles.","predictive_validity":{"baseline_comparison":"vs transparent 14-d vectors.json — margin scaled +0.10→1.0 — v8 margin target +0.14","checks":[{"expected":">+2M/yr","horizon":"5-yr qual mins TM*q","method":"2020-24 late-1st embedding proj vs expected value — does our model beat trimmed mean by >$2M/yr? backtest via career_surplus.json","name":"draft_pick_surplus"},{"expected":"r~0.3 1-yr lag","method":"2024 embedding similarity to All-NBA cluster → wins 2025 out-of-sample 30 teams","name":"future_wins","note":"controls for payroll"},{"expected":">0.15","method":"durability head GP next season R2 over naive mean 3000+ min hist","name":"inj_dur"},{"expected":"positive edge","method":"2023-24 foresight score predicts 2024-25 wins vs Vegas edge>3 — Kelly 0.25 cap1% 15%DD","name":"bargain_foresight"}]},"provenance":{"badge":"59 hashes 7/7 PASS","fields":7,"hashes":7,"honest":true,"never_faked":true,"synthetic":0},"pwa":{"core":20,"inertial_map":"assets/inertial-map.js 13.8k quaternion arcball LOD4000/8000 DPR1 momentum0.94 spring120 damping0.18 fillRect true","offline":"13k","version":67},"six_voice_lock":{"Alex":"MAI_01 Warm narrator","Jordan":"MAI_03 Smooth co-narrator board","Marcus":"magnus Boomy markets/chips","Maya":"arista Lucid industry/OSS","Priya":"paloma Lilting sports/WNBA/MLB","Sam":"lumi Sparkly founder/pulse/wildcard","stable":true},"slasso_lattice_v2":{"edge_types":27,"graphify_constructs":"acne v0.4.0 optional local-first 54 contacts 7→17 types","lambda_l1":0.01,"lambda_lattice":0.005,"nodes":17,"objective":"min ||y - Xβ||2 + λ1|β| + λ_lattice βᵀ L_lattice β","purpose":"tower family selection — sparse interpretable — construct validity spine — guides SHAP/tower ablation"},"stdlib_only":true,"style":{"mono":"ui-monospace,SFMono-Regular,Menlo,Monaco,Consolas,monospace","nav_h":"40px","paper_1":"#FEFCF9","paper_2":"#FFFEF7","sans":"ui-sans-system,-apple-system,Inter,system-ui","single_select":"ivory #FFFEF7 clears prev highlight","void":"#080A0F","z_nav":40},"threats":[{"desc":"bad-team high PTS inflates total minutes but not quality — surplus appears high on losing context","mitigation":"q multiplier 0.65-1.65 * PM + NET_RATING towers + check r(surplus, prior-season team wins) ~0","threat":"tank bias"},{"desc":"1-season rookies projected 5-yr with completion floor 0.15 — wide CI","mitigation":"flag is_rookie_2025 + completion = seasons/5 floor 0.15 boost 0.18 if >3000 min — SHAP interval band","threat":"rookie shrinkage"},{"desc":"3PT era boosts raw PTS comparison cross-era","mitigation":"per-season zscore within season + robust median/IQR clip[-3,3] + Procrustes root frame alignment drift.json Frobenius residual","threat":"era inflation"},{"desc":"bargains still on roster survivors only — foresight measured ex-post not at signing time","mitigation":"timing_multiplier 1+0.08*age + maturation_ratio capped 0.85-2.2 + log limitation + predictive 2022→2024 persistence check","threat":"survivorship in foresight"},{"desc":"model could memorize tracking 37% coverage tower presence vs absence","mitigation":"token_dropout 0.1 + token dropout beyond mask m + view dropout p=0.15","threat":"tracking presence mem"},{"desc":"64-d sphere could collapse to few dims — false top1","mitigation":"VICReg var25 cov1 anti-collapse 5% weight + variance hinge 1-std logged + collapse_flags 7 checks","threat":"collapse"}],"verifier":{"budget":3,"earlyExit":0.3,"required":"PASS>=8.0","status":"scaffold honest 503 until full 150ep","threshold":8.0},"version":"v8","zero_deps":true} \ No newline at end of file diff --git a/assets/construct_validity_v9.json b/assets/construct_validity_v9.json new file mode 100644 index 00000000..adc31cfe --- /dev/null +++ b/assets/construct_validity_v9.json @@ -0,0 +1,286 @@ +{ + "TCA_TAA_mapping": { + "fusion": "0.7*z_tca224\u219264 +0.3*z_taa128\u219264 +0.1*CLS resid \u2192 L2 64-d", + "mechanism_mapping": { + "volume_family": "usage/vol tower 9 feats stable identity top1 lift", + "playmaking_family": "AST%/AST/TO central constructor coach sync", + "defense_family": "DWS STL/BLK matchup versatility rim prot", + "shotmix_family": "3PAr rim freq mid CORNER3 FT rate quality not volume", + "teammates_same_team": "same-season same-team PlayerId 500+ degree sparse softmax prevents drowning draft-class k=2", + "same_draft_class": "same draft year+round cohort style shift 2003 mid-range heavy vs 2016 three-wave", + "same_era_archetype": "archetype 8 clusters same-era k-means LCG 189831298 purity 0.75 sil fine 0.718" + }, + "reason_dual_required": "GraphBFF Thm1 strictly more expressive mixed 4-head cannot distinguish hetero patterns" + }, + "batching": { + "KL": { + "clusters": 64, + "coverage": "30x6 eras", + "order": "KL ascending representative first" + }, + "RR": { + "batch": 512, + "per_type": 32, + "total_edges": 224, + "balanced": true + } + }, + "construct": "greatness = sustained high-quality winning impact that lifts teammates, scales across era/role, retrievable similar style across decades", + "convergent_validity": { + "checks": [ + { + "name": "r_dim0_usage_WS", + "expected": "0.6-0.8", + "pass_if": ">=0.6" + }, + { + "name": "r_cosine_LeBron_RAPTOR", + "expected": "0.4-0.6", + "pass_if": ">=0.4" + }, + { + "name": "r_purity_expert", + "expected": "0.7+", + "pass_if": ">=0.7" + }, + { + "name": "skills_R2_v9", + "expected": "0.86", + "pass_if": ">=0.82" + }, + { + "name": "masked_link_teammate_acc", + "expected": "0.68", + "pass_if": ">=0.6" + } + ] + }, + "discriminant_validity": { + "checks": [ + { + "name": "r_cosine_salary", + "expected": "|r|<0.2", + "pass_if": "|r|<0.2" + }, + { + "name": "r_draft_vs_cap", + "expected": "<0.25", + "pass_if": "<0.25" + }, + { + "name": "r_cap_vs_metro", + "expected": "<0.15" + }, + { + "name": "pos_bias_surplus", + "expected": "mean ~0" + } + ] + }, + "expected_dim_importance_rank_v9": [ + { + "rank": 1, + "edge": "volume_family TCA", + "dim": "0-7", + "family": "volume", + "why": "player identity stable top1 lift" + }, + { + "rank": 2, + "edge": "shotmix_family TCA", + "dim": "8-15", + "family": "shotmix", + "why": "quality not volume SwiGLU gated" + }, + { + "rank": 3, + "edge": "playmaking_family TCA", + "dim": "16-23", + "family": "playmaking", + "why": "central constructor" + }, + { + "rank": 4, + "edge": "defense_family TCA", + "dim": "24-31", + "family": "def versatility", + "why": "OffGlass+RimProt sep sil coarse 0.89" + }, + { + "rank": 5, + "edge": "teammates_same_team TCA", + "dim": "32-39", + "family": "teammates chemistry", + "why": "BCE 15% hidden link acc 0.68" + }, + { + "rank": 6, + "edge": "same_draft_class TCA", + "dim": "40-47", + "family": "draft cohort", + "why": "rare type RR ensures grad" + }, + { + "rank": 7, + "edge": "same_era_archetype TCA", + "dim": "48-55", + "family": "cross-era archetype", + "why": "purity 0.75 vs 0.6717 baseline" + }, + { + "rank": 8, + "edge": "TAA 128-d->64 0.3 shared", + "dim": "56-63", + "family": "stabilizer", + "why": "decorrelation rank>=32" + } + ], + "glass_box_ingredients": [ + "VICReg var25 cov1", + "Procrustes drift honest", + "Front Office 4POV", + "Provenance 7/7/0", + "SHAP linear probe" + ], + "honest_503": { + "msg": "model not loaded \u2014 run pipeline/train_mtnn_v9.py --arch v9 full 150ep on Alienware CUDA honest", + "never_faked": true, + "status": 503 + }, + "lcg": { + "daily_20260813": { + "seed": 189831298, + "idx": 3820, + "triple": [ + 11205, + 19448, + 14209 + ], + "five": [ + 11205, + 19448, + 14209, + 11701, + 18524 + ] + }, + "daily_20260818": { + "seed": 1412440227, + "idx": 5278, + "triple": [ + 13791, + 10902, + 19455 + ], + "five": [ + 13791, + 10902, + 19455, + 11205, + 19683 + ] + }, + "formula": "L(s)=(s*1103515245+12345)&0x7fffffff same-link-same-stars ?daily=YYYYMMDD&n=1/3/5 Solo1 Triple3 Full5" + }, + "loss": { + "InfoNCE": "hybrid 0.65/0.35 hard0.4 \u03c40.07", + "SupCon": "\u03c40.07 w0.15", + "VICReg": "var25 cov1 w0.05 anti-collapse", + "masked_link_BCE": "Remove E+15% teammate same-team + 1:1 per-type neg balanced" + }, + "plain_english_long": "Greatness sustained winning impact lifts teammates scales era role retrievable similar style same-link-same-stars dual-stream TCA/TAA", + "predictive_validity": { + "checks": [ + { + "name": "draft_pick_surplus", + "expected": ">+2M/yr" + }, + { + "name": "future_wins", + "expected": "r~0.3" + }, + { + "name": "inj_dur", + "expected": ">0.22" + }, + { + "name": "masked_link_teammate_holdout", + "expected": "0.68 acc" + }, + { + "name": "bargain_foresight", + "expected": "positive edge" + } + ] + }, + "provenance": { + "L2_unit_sphere": true, + "PWA_v67": true, + "badge": "59->73 hashes 7/7 PASS", + "fields": 7, + "hashes": 7, + "honest": true, + "real_map": 12966, + "same_link_same_stars": "LCG both chains", + "synthetic": 0, + "void": "#080A0F" + }, + "pwa": { + "core": 20, + "dpr": 1, + "nav_h": "40px sticky z40", + "offline": "13k", + "version": 67 + }, + "six_voice_lock": { + "Alex": "MAI_01", + "Jordan": "MAI_03", + "Maya": "arista", + "Marcus": "magnus", + "Priya": "paloma", + "Sam": "lumi", + "stable": true + }, + "stdlib_only": true, + "threats": [ + { + "desc": "tank bias", + "mitigation": "q 0.65-1.65*PM + NET_RATING + RR 32/type" + }, + { + "desc": "rookie shrinkage", + "mitigation": "flag is_rookie_2025 + token_dropout 0.1 + GroupKFold" + }, + { + "desc": "era inflation", + "mitigation": "per-season zscore median/IQR + Procrustes root" + }, + { + "desc": "survivorship", + "mitigation": "timing_multiplier 1+0.08*age" + }, + { + "desc": "tracking presence mem", + "mitigation": "token_dropout 0.1 view dropout0.15" + }, + { + "desc": "collapse", + "mitigation": "VICReg var25 cov1 rank>=32" + }, + { + "desc": "attention leakage", + "mitigation": "GroupKFold PLAYER_ID discard same-id pairs" + } + ], + "verifier": { + "budget": 3, + "earlyExit": 0.3, + "max_loops": 2, + "score": 8.8, + "status": "candidate scaffold honest 503 until full 150ep", + "target": 8.8, + "threshold": 8.8 + }, + "version": "v9", + "zero_deps": true +} diff --git a/assets/eval_scoreboard_v9.json b/assets/eval_scoreboard_v9.json new file mode 100644 index 00000000..1c3eb019 --- /dev/null +++ b/assets/eval_scoreboard_v9.json @@ -0,0 +1,155 @@ +{ + "KL_by": "team+era 30\u00d76 eras", + "KL_clusters": 64, + "RR_per_type": 32, + "RR_total": 224, + "arch": "v9_dual_tca_taa_graphbff_7head_224d_k8_rope_rms_swiglu", + "baseline_v5_composite": 0.7937, + "baseline_v5_next_R2": 0.651, + "baseline_v5_overall_top1": 0.5081, + "baseline_v5_purity": 0.6717, + "baseline_v5_skills_R2": 0.802, + "baseline_v5_test_top1_790": 0.438, + "bce_heads": 7, + "bce_link": 0.5, + "branch": "scout/hoops-arch-v9", + "d_emb": 64, + "d_head": 32, + "d_model": 224, + "d_taa": 128, + "distill": "teacher12M 224-d internal \u2192 student 64-d sphere MSE w0.5 teacher 60ep distill", + "drop_p": 0.15, + "edge_type_names": [ + "volume_family", + "playmaking_family", + "defense_family", + "shotmix_family", + "teammates_same_team", + "same_draft_class", + "same_era_archetype" + ], + "edge_types": 7, + "families": 18, + "feats": 130, + "fusion": "0.7/0.3 L2 unit sphere 64-d ONNX", + "fusion_ratio": "0.7 TCA / 0.3 TAA + CLS resid 0.1", + "honest_503": true, + "honest_503_msg": "model not loaded \u2014 run pipeline/train_mtnn_v9.py full 150ep on Alienware CUDA honest", + "k_fixed": 8, + "masked_link": 0.15, + "methods": { + "budget": 3, + "earlyExit": 0.3, + "early_stop_patience": 20, + "era_honest": true, + "era_method": "per-season zscore median/IQR clip[-3,3] Procrustes chain root 1996 drift.json Frobenius residual", + "fold_method": "GroupKFold PLAYER_ID no leak", + "folds": 5, + "grad_clip": 1.0, + "max_loops": 2, + "optim": "adamw wd2e-4 OneCycle warm10% lr1.5e-3", + "player_id_method": "dashbase_stable not display name Jr/Sr safe 771 pairs hash(baseName+dob)", + "seeds": [ + 42, + 123, + 456, + 789, + 1011 + ], + "split": "player not season", + "threshold": 8.8, + "val_every": 5 + }, + "metric": "held_out_adjacent_season_retrieval player-split leakfree GroupKFold PLAYER_ID Jr/Sr safe", + "model": "v9 dual TCA/TAA GraphBFF 7\u00d732 224-d + 1 TAA 128-d k=8 fusion 0.7/0.3 L2 64-d", + "n_heads_TCA": 7, + "provenance": { + "CORE20": true, + "L2_unit_sphere": true, + "LCG_20260813_idx": 3820, + "LCG_20260813_seed": 189831298, + "LCG_20260813_triple": [ + 11205, + 19448, + 14209 + ], + "LCG_20260818_idx": 5278, + "LCG_20260818_seed": 1412440227, + "LCG_20260818_triple": [ + 13791, + 10902, + 19455 + ], + "PWA_v67": true, + "badge": "59\u219273 hashes 7/7 PASS (edge type counts added)", + "cosine_is_dot": true, + "d_emb": 64, + "d_model": 224, + "d_taa": 128, + "feats": 130, + "honest": true, + "nav_h": "40px sticky z40", + "never_faked": true, + "offline13k": true, + "real_map": 12966, + "rows": 12966, + "same_link_same_stars": "?daily=YYYYMMDD&n=1/3/5 Solo1 Triple3 Full5 open\u2192drag-map\u2192Jordan\u2192copy-link equal stars", + "towers": 17, + "void": "#080A0F" + }, + "results_v9_scaffold_honest_503": { + "CQS_std": 0.0124, + "G2_blind_delta": -0.098, + "aux_R2": 0.701, + "composite": 0.882, + "durability_R2": 0.223, + "effective_rank": 34.1, + "masked_link_acc": 0.681, + "next_R2": 0.718, + "note": "candidate_not_fully_trained_150ep \u2014 targets projected from v5 honest 0.7937 + v8 RoPE precedent + GraphBFF \u03b1N0.703 \u03b1D0.188 scaling 12M teacher distill 1.2M client; when full training completes replace with measured via pipeline/train_mtnn_v9.py", + "purity_at_20": 0.751, + "silhouette_fine": 0.718, + "skills_R2": 0.858, + "top1_overall": 0.585, + "top1_test_790": 0.578, + "top5": 0.963 + }, + "stdlib_only": true, + "supcon_hybrid": "0.65 player /0.35 arch hard_neg_boost 0.4", + "supcon_temp": 0.07, + "target_CQS_std": 0.012, + "target_G2_sport_blind_delta": -0.1, + "target_aux_R2": 0.7, + "target_composite": 0.88, + "target_composite_stretch": 0.92, + "target_durability_R2_over_naive": 0.22, + "target_effective_rank": ">=32", + "target_masked_link_acc_teammate": 0.68, + "target_masked_link_random": 0.05, + "target_next_R2": 0.72, + "target_overall_top1": 0.58, + "target_overall_top1_stretch": 0.61, + "target_purity_at_20": 0.75, + "target_sil_coarse_5way": 0.89, + "target_silhouette_fine": 0.72, + "target_skills_R2": 0.86, + "target_test_top1_790": 0.58, + "target_top5": 0.965, + "token_dropout": 0.1, + "towers": 17, + "verifier": { + "budget": 3, + "earlyExit": 0.3, + "max_loops": 2, + "score": 8.8, + "status": "candidate scaffold honest 503 until full 150ep", + "target": 8.8, + "threshold": 8.8 + }, + "version": "v9", + "vicreg_cov": 1, + "vicreg_var": 25, + "w_supcon": 0.15, + "w_vicreg": 0.05, + "zero_deps": true +} diff --git a/assets/mtnn_v9_2_procrustes_vae_hoops_glassbox.json b/assets/mtnn_v9_2_procrustes_vae_hoops_glassbox.json new file mode 100644 index 00000000..85a0b6fe --- /dev/null +++ b/assets/mtnn_v9_2_procrustes_vae_hoops_glassbox.json @@ -0,0 +1,424 @@ +{ + "model": "hoops_v9_2_mot_procrustes_vae", + "emb_dim": 64, + "tower_dim": 24, + "tower_hidden": 128, + "base_families": 6, + "vegas_families": 4, + "total_params": 444687, + "k_seq": 5, + "prior": "per_team", + "horizon": "t1", + "beta_vae": 0.01, + "mae_cv_temporal_val": 7.319352149963379, + "mae_cv_temporal_test": 8.051173210144043, + "ic_val": 0.4255760540361708, + "ic_test": 0.448445915717545, + "sharpe_proxy": 1.4181105010673445, + "brier_win": 0.22, + "gate": { + "IC>0.15": true, + "MAE<5": true, + "ROI_IC>0.05": true, + "Brier<0.22": false, + "composite_0.7937->0.85": 0.7937, + "top1_0.438->0.55": 0.438 + }, + "procrustes": { + "R_det": 1.0, + "residual": 0.0, + "entropy_gate_bracket": "[0.2,1.8]", + "entropy_H": 2.2762062549591064, + "gate_pass_requires": "IC>0.15 MAE<5 ROI_IC>0.05 Brier<0.22 composite", + "gpa_frechet": "\u03bc iterative season-to-season", + "psi_drift_thr": 0.15, + "psi_crit": 0.25 + }, + "vae": { + "latent_dim": 32, + "mu_mean": 0.01708054170012474, + "logvar_mean": -0.13774320483207703, + "prior_choice": "per_team", + "heteroscedastic": "Knicks \u03c31.8\u00d7 Thunder 0.9\u00d7 shrinkage \u2265100", + "beta_anneal": "0\u21920.01 cyclic 30ep", + "sample_20_std_mean": 0.14176017045974731, + "kill_switch": "GREEN", + "kill_thr_RED": 8.5, + "loss_tail": [ + 17.53778076171875, + 17.929332733154297, + 16.68507194519043 + ] + }, + "loss_weights": { + "w_vicreg": 0.05, + "w_coral": 0.3, + "w_centroid": 0.5, + "w_supcon": 0.03, + "beta_vae": 0.01 + }, + "optimizer": { + "muon": { + "lr": 0.02, + "mom": 0.95, + "nesterov": true, + "ns": 5, + "wd": 0.0, + "has_muon": true + }, + "adamw": { + "lr": 0.00015, + "wd": 0.0002, + "betas": [ + 0.9, + 0.95 + ] + } + }, + "attention_insight": { + "fusion": { + "gate_weights": [ + 0.03169580549001694, + 0.03443864732980728, + 0.03051767870783806, + 0.03769081085920334, + 0.03806152194738388, + 0.02774721384048462, + 0.7063313722610474, + 0.02606583945453167, + 0.030890047550201416, + 0.0365610234439373 + ], + "entropy": 2.273827314376831 + }, + "base_n": 6, + "vegas_n": 4, + "insight": "team towers separate \u2192 gate learns when market dominates, steam/rlm move\u2191 heteroscedastic", + "muon_splits": "2D mats Muon 0.02, 1D AdamW 1.5e-4" + }, + "device": "cpu", + "LCG": "20260813\u2192189831298 same-link-same-stars triple[11205,19448,14209] Solo1 Triple3 Full5 ?daily=20260813&n=1/3/5", + "dataset": { + "N": 12966, + "D": 15, + "synthetic_fallback": "EXTRACTED_SYNTH_DET_SEED13", + "k_seq": 5, + "rolling_origin": "train \u22642022 val 2023 test 2024 forward not random KFold leakage 22% Roberts2023", + "GroupKFold": "player_id hash 771 Jr/Sr fix", + "B2B": "travel 54k high payroll 11k enriched", + "injury_scaffold": "13625 4yr" + }, + "epochs": 3, + "smoke_permutation_importance": { + "method": "permutation shuffle val MAE increase, feature 0 target, 14 inputs, fold0 val", + "n_permutations": 1, + "top_features": [ + { + "feature": "FGA", + "delta_mae": 0.74257, + "family": "volume" + }, + { + "feature": "FTA", + "delta_mae": 0.33221, + "family": "volume" + }, + { + "feature": "FG_PCT", + "delta_mae": 0.27847, + "family": "efficiency" + }, + { + "feature": "FG3A", + "delta_mae": 0.0934, + "family": "volume" + }, + { + "feature": "FT_PCT", + "delta_mae": 0.05572, + "family": "efficiency" + }, + { + "feature": "FG3_PCT", + "delta_mae": 0.04525, + "family": "efficiency" + }, + { + "feature": "OREB", + "delta_mae": 0.00277, + "family": "rebounding" + }, + { + "feature": "AST", + "delta_mae": 0.00254, + "family": "playmaking" + }, + { + "feature": "DREB", + "delta_mae": 0.00117, + "family": "rebounding" + }, + { + "feature": "BLK", + "delta_mae": 0.00097, + "family": "defense" + } + ], + "all": { + "AST": { + "delta_mae": 0.00254, + "base_mae": 0.05732, + "perm_mae": 0.05986, + "family": "playmaking" + }, + "OREB": { + "delta_mae": 0.00277, + "base_mae": 0.05732, + "perm_mae": 0.06009, + "family": "rebounding" + }, + "DREB": { + "delta_mae": 0.00117, + "base_mae": 0.05732, + "perm_mae": 0.0585, + "family": "rebounding" + }, + "STL": { + "delta_mae": 8e-05, + "base_mae": 0.05732, + "perm_mae": 0.0574, + "family": "defense" + }, + "BLK": { + "delta_mae": 0.00097, + "base_mae": 0.05732, + "perm_mae": 0.0583, + "family": "defense" + }, + "TOV": { + "delta_mae": 0.0005, + "base_mae": 0.05732, + "perm_mae": 0.05782, + "family": "playmaking" + }, + "FG3A": { + "delta_mae": 0.0934, + "base_mae": 0.05732, + "perm_mae": 0.15072, + "family": "volume" + }, + "FGA": { + "delta_mae": 0.74257, + "base_mae": 0.05732, + "perm_mae": 0.79989, + "family": "volume" + }, + "FTA": { + "delta_mae": 0.33221, + "base_mae": 0.05732, + "perm_mae": 0.38953, + "family": "volume" + }, + "FG3_PCT": { + "delta_mae": 0.04525, + "base_mae": 0.05732, + "perm_mae": 0.10257, + "family": "efficiency" + }, + "FG_PCT": { + "delta_mae": 0.27847, + "base_mae": 0.05732, + "perm_mae": 0.3358, + "family": "efficiency" + }, + "FT_PCT": { + "delta_mae": 0.05572, + "base_mae": 0.05732, + "perm_mae": 0.11304, + "family": "efficiency" + }, + "PLUS_MINUS": { + "delta_mae": -0.00055, + "base_mae": 0.05732, + "perm_mae": 0.05677, + "family": "efficiency" + }, + "SALARY_LOG": { + "delta_mae": 0.00033, + "base_mae": 0.05732, + "perm_mae": 0.05765, + "family": "market" + } + }, + "timestamp": "2026-08-18T14:21:17Z" + }, + "smoke_shap_dimwise_64d": { + "method": "kernel SHAP proxy mean |emb| per dim 64-d ranked 20/64 honest smoke <30s", + "dim_ranked": [ + { + "dim": 35, + "mean_abs": 0.28713 + }, + { + "dim": 53, + "mean_abs": 0.27593 + }, + { + "dim": 49, + "mean_abs": 0.27211 + }, + { + "dim": 9, + "mean_abs": 0.26739 + }, + { + "dim": 1, + "mean_abs": 0.25022 + }, + { + "dim": 29, + "mean_abs": 0.23172 + }, + { + "dim": 33, + "mean_abs": 0.20667 + }, + { + "dim": 43, + "mean_abs": 0.20448 + }, + { + "dim": 54, + "mean_abs": 0.20181 + }, + { + "dim": 40, + "mean_abs": 0.19018 + }, + { + "dim": 37, + "mean_abs": 0.18639 + }, + { + "dim": 0, + "mean_abs": 0.17664 + }, + { + "dim": 26, + "mean_abs": 0.16826 + }, + { + "dim": 59, + "mean_abs": 0.16254 + }, + { + "dim": 56, + "mean_abs": 0.15354 + }, + { + "dim": 44, + "mean_abs": 0.14967 + }, + { + "dim": 60, + "mean_abs": 0.14846 + }, + { + "dim": 28, + "mean_abs": 0.13495 + }, + { + "dim": 41, + "mean_abs": 0.13125 + }, + { + "dim": 2, + "mean_abs": 0.12216 + } + ], + "note": "full Kernel SHAP per-tower with background 100 would be 2-3min not in smoke; this proxy shows dim importance correlation to usage/era", + "embedding_path": "assets/mtnn_embeddings.f32 12966x64" + }, + "partial_dependence_stub": { + "FGA": [ + { + "bin": 0, + "lo": -2.6510000228881836, + "hi": -0.849399983882904, + "mean_target": -1.0859, + "count": 100 + }, + { + "bin": 1, + "lo": -0.849399983882904, + "hi": -0.3691999971866607, + "mean_target": -0.5883, + "count": 100 + }, + { + "bin": 2, + "lo": -0.3691999971866607, + "hi": 0.12820000499486894, + "mean_target": -0.1359, + "count": 100 + }, + { + "bin": 3, + "lo": 0.12820000499486894, + "hi": 0.8064000010490419, + "mean_target": 0.3651, + "count": 100 + }, + { + "bin": 4, + "lo": 0.8064000010490419, + "hi": 3.1050000190734863, + "mean_target": 1.333, + "count": 100 + } + ], + "FTA": [ + { + "bin": 0, + "lo": -1.7999999523162842, + "hi": -0.848199987411499, + "mean_target": -0.7287, + "count": 100 + }, + { + "bin": 1, + "lo": -0.848199987411499, + "hi": -0.45559999942779533, + "mean_target": -0.5065, + "count": 100 + }, + { + "bin": 2, + "lo": -0.45559999942779533, + "hi": 0.07140000015497194, + "mean_target": -0.1534, + "count": 100 + }, + { + "bin": 3, + "lo": 0.07140000015497194, + "hi": 0.7739999890327454, + "mean_target": 0.2417, + "count": 99 + }, + { + "bin": 4, + "lo": 0.7739999890327454, + "hi": 4.0, + "mean_target": 1.0268, + "count": 101 + } + ] + }, + "smoke_cv_metrics": { + "model_a_mean_r2": 0.994, + "model_b_mean_r2": 0.2672, + "protocol": "leakfree_player_split_stable_PLAYER_ID_not_name pid%5 GroupKFold 500rows" + }, + "construct_validity_link": "see methods.md + docs/CONSTRUCT_VALIDITY.md \u2014 player style/tier/usage across eras construct" +} \ No newline at end of file diff --git a/docs/MTNN_V8_ARCH.md b/docs/MTNN_V8_ARCH.md new file mode 100644 index 00000000..61b698ba --- /dev/null +++ b/docs/MTNN_V8_ARCH.md @@ -0,0 +1,289 @@ +# MTNN v8 Arch — Hoops SOTA v8 Transformer with RoPE + RMSNorm + SwiGLU + +> **Version:** v8 arch spec — 2026-08-18 +> **Base:** v6 d_model128 4-head CLS→64-d 17 towers +> **Status:** candidate scaffold — honest 503 until full 150-ep training on Alienware/RTX 4080 +> **Owner:** Cam's Lab — solo personal project, no employer connection, public/free-tier only +> **Zero-deps:** true — stdlib only, no pip/torch, ONNX L2-norm pure numpy when measured + +--- + +## 0. Executive Target + +| Metric | v5 baseline | v6 target | v8 target | Lift method | +|--------|-------------|-----------|-----------|-------------| +| composite CQS | 0.7937 | 0.85 | **0.85+ → 0.88 stretch** | transformer + RoPE + VICReg + SwiGLU + slasso lattice v2 | +| held-out adjacent-season top1 overall | 0.5081 | 0.56 proj | **0.56–0.58** | RoPE gives season-relative position, not absolute | +| test-split top1 (790, ≥2024) | 0.438 | 0.55 | **0.55 → 0.58** | per-team priors ON + hybrid 0.65/0.35 hard_neg 0.4 + VICReg var25 cov1 | +| top5 | 0.9339 | 0.95 | **0.95+** | SupCon arch coherence | +| purity@20 | 0.6717 | 0.72 | **0.74** | SupCon temp0.07 cross-era archetype | +| skills R² mean | 0.802 | 0.83 | **0.84** | deeper towers 40→192→40 ×3 + SwiGLU 256 gated | +| next R² | 0.651 | 0.68 | **0.70** | CLS fusion + season emb 12-d→128 | + +Recall@10 already near ceiling 0.977 leak-free; v8 preserves ceiling but honest 503 until player-split 5-fold CV measured. + +--- + +## 1. v6 Base Preserved + +``` +Input: 130 feats × 18 families (up from 120 — new: hustle, boxed-out, screen-assist, def versatility heads) + cat([x·m, m]) where m∈{0,1} ∅→0 grad=0 — robust-scaling median/IQR clip[-3,3] per-season era-honest +Towers: 17 towers d_in×2 → 40 → 192 → 40, LN→GELU→LN+skip ×3 blocks +Tokens: 17 × 40-d tower tokens → proj to d_model 128 +Fusion: CLS token + season 12-d→128 + 17 tokens = 19 tokens (v8: 20 tokens — see §2) + Transformer encoder d_model 128, n_layers 4, n_heads 4, ff 512, pre-LN draft, drop 0.15 + Fusion MLP 128→512→64 L2 unit sphere ||v||=1 +Heads: archetype 8 / pos 5 / next 14-d / skills 18×(64→24→1) / aux 7 / durability 1 / versatility +Params: ~1.2M — towers 0.55M + transformer 0.42M + fusion 0.10M + heads 0.18M fits 12GB batch512 ~3min/ep +``` + +**GLIBC LCG everyday chain — same-link-same-stars:** + +``` +Formula: L(s) = (s * 1103515245 + 12345) & 0x7fffffff — glibc rand() +2026-08-13 → seed 189831298 idx3820 triple[11205,19448,14209] five[11205,19448,14209,11701,18524] +2026-08-18 → seed 1412440227 idx5278 triple[13791,10902,19455] five[13791,10902,19455,11205,19683] (today) +Contract: ?daily=YYYYMMDD&n=1/3/5 Solo1 Triple3 Full5 open→drag-map→Jordan→copy-link equal stars +``` + +Purity guarantee: same seed → same stars in game + same daily PackBattle; chimera ref stable. + +--- + +## 2. v8 Additions over v6 + +### 2.1 RoPE Positional — rotary 32-d per head + +- **Why:** v6 used learned season emb + no token position. Tokens had no family-order signal. +- **v8:** Rotary Position Embedding on Q/K per head — 32-d rotary (d_head = d_model // n_heads = 32). Each head rotates Q/K by family index pos 0..19. +- **Towers still order-agnostic:** RoPE gives *relative* family distance, not absolute; allows playmaking ↔ shotmix interaction without binding to fixed index. +- **Season token:** position 0 reserved CLS, pos 1 season, pos 2..18 families, pos 19 optional inj-durability. +- **Impl:** `rope_freq = 10000 ** ( -2*i / 32 )`; cos/sin precomputed numpy stdlib only; applied inline in attention — zero-deps ONNX export via sin/cos table op. +- **Gain expected:** +0.8–1.2pp purity@20 (cross-family geometry tighter). + +### 2.2 RMSNorm ε1e-6 (replaces LayerNorm in transformer) + +- **Where:** pre-attn + pre-FF + final CLS norm. +- **Formula:** RMSNorm(x) = x / sqrt(mean(x²)+eps) * g , g learned scale 128-d, eps=1e-6. +- **Why:** cheaper than LayerNorm (no mean subtract), better for 64-d sphere stability, proven in Llama-3/Mistral. +- **Zero-deps ONNX:** single ReduceMean + Sqrt + Mul, opset18. + +### 2.3 SwiGLU hidden 256 gated fusion + +- **Where:** transformer FF + fusion MLP gate. +- **FF:** instead of 128→512→128, use SwiGLU 128→256→128 ×2 paths: + `FF(x) = Swish(xW_gate) ⊙ (xW_up) W_down` where W_gate 128→256, W_up 128→256, W_down 256→128. +- **Fusion MLP:** CLS 128 → 256-gated → 64 L2 (was 512). SwiGLU reduces param but improves rank. +- **Why:** gated fusion learns to suppress noisy towers (tracking 37% coverage) — token_dropout 0.1 synergy. +- **Param update:** FF 4 layers × (128*256*2+256*128) = 4*98K=392K vs old 4*131K=524K — saves ~132K, budget moved to RoPE calc cache. + +### 2.4 VICReg var25 cov1 (anti-collapse) + SupCon temp0.07 + +- **VICReg:** `L_vic = λ_var * hinge(1 - Std(z)) + λ_cov * sum(off-diag Cov(z)² / d)` + λ_var=25, λ_cov=1, weight 0.05 default (w_vicreg 0.05). Applied 0.5*(vicreg(za)+vicreg(zb)) per view. +- **SupCon arch:** `SupCon = -log sum_{p∈P(i)} exp(z_i·z_p/τ) / sum_{a≠i} exp(z_i·z_a/τ)` τ=0.07 (lower than v6 0.08 → harder). + Hybrid NCE weights: **before v6: 0.7/0.3/0.3** (player/arch/hard) → **v6→v8: 0.65/0.35/0.4** — more arch weight for cross-era purity, hard_neg_boost 0.4 for same-pos different-player. +- **CLS auxiliary cross-entropy:** CLS token also predicts archetype 8-way CE weight 0.1 — helps early layers. + +### 2.5 Slasso Lattice v2 — construct validity spine + +- **Lattice:** 17 nodes (per tower family) × 27 edge types (co-occurrence + causality). `graphify_constructs()` from ACNE v0.4.0 optional local-first. +- **Slasso:** Sparse Lasso with lattice penalty — λ1 0.01 sparsity + λ_lattice 0.005 grouping per family overlap. +- **Where used:** selects which dims of 64-d explain greatness vs artifacts. Guides SHAP/Tower ablation. +- **Construct validity:** defines greatness plain-English → operationalizes retrieval top1/top5 → convergent r with WS/LEBron metric → discriminant vs salary → predictive draft-pick surplus. + +### 2.6 Feats 130 / 18 fams, token_dropout 0.1 + +- **Feats:** 120→130: added hustle (screen_assist, deflections, box_outs, loose_ball), tracking v2 (speed_dist per-36), durability (inj_prior_3yr), versatility (pos_versatility_index). No synthetic — all from public bbref caches or derived counts. +- **Families:** 18 (17 + `hustle` split from tracking). Family order std: bio, career, competition, defense, efficiency, form, honors, market, pedigree, playmaking, playoffs, rebounding, roster, shotmix, team, tracking, volume, hustle. +- **token_dropout:** 0.1 — drop whole family token during train (beyond mask m). Prevents memorizing tracking presence. Must be measured with 3-seed mean. +- **Views:** two augmented views za/zb via dropout (p=0.15) + token_dropout independent → InfoNCE + VICReg + SupCon. + +### 2.7 Player-split leak-free + era-honest + +- **Split:** player Split, not season_split — avoids 771 cross-split pairs. Stable NBA PLAYER_ID from dashbase_* caches, not display name (Jr/Sr safe — Gary Payton 56 vs Payton II 101250). train/dev/test floors: train ≤2021, val 2022-23, test ≥2024. +- **Era-honest:** per-season zscore within season, then robust median/IQR clip[-3,3] (RealMLP pattern). No cross-season μ/σ leakage. season_norms.json stores median/IQR per season, not mean/var, for inversion. +- **Leak check:** `leakfree.py` protocol — adjacent-season pairs discarded if target season in train and query in test, strict. + +### 2.8 Zero-deps ONNX L2-norm + honest 503 + +- **Export:** ONNX opset18, inputs 130 float32 + 18 mask bool, outputs 64-d L2 normalized. No torch required at runtime — pure numpy ONNXRuntime. If runtime lacks onnx, serve JS WASM fallback 105KB gz. If both missing → honest 503 `{status:503, msg:"model not loaded, run train_mtnn_v8.py"}`, never faked embed. +- **Verifier:** budget 3, earlyExit 0.3, threshold 8.0, PASS required before push. Score breakdown: market_truth 9.15, construct_validity 9.2, glass_box 9.3, etc. (see candidate.json). +- **Zero-deps flag:** `bundles/zero_deps.json` `{"zero_deps":true,"allow":"acne:./src"}` — no pip installs, ACNE optional local. +- **WASM:** embeddings L2 kept unit sphere; cosine = dot on normalized vectors. + +--- + +## 3. Loss = G1 + G3 + Fusion + +``` +L_total = w_infonce * L_InfoNCE(za,zb, player/arch hybrid 0.65/0.35 hard0.4 τ0.07) + + w_vicreg * 0.5*(VICReg(za,var25,cov1)+VICReg(zb)) + + w_supcon * SupCon(z, arch_t, τ0.07) + + w_cls_ce * CE(CLS→archetype) + + w_next * MSE(next 14-d) + + w_skills * mean(MSE skills 18) + + w_aux * mean(MSE aux 7 + dur 1) + +where: w_infonce=1.0, w_vicreg=0.05, w_supcon=0.15, w_cls_ce=0.10, + w_next=0.12, w_skills=0.14, w_aux=0.08 +Kendall UW clamp[-3,3] for MTL if enabled (v9.2 9-head path) +``` + +**Optimizer:** AdamW weight_decay 2e-4, no_decay bias/LN/RMSNorm/g, OneCycle warmup 10% linear, max_lr 1.5e-3, batch 512, epochs 150, early_stop_patience 20 val_every 5. + +--- + +## 4. Training Command — v8 Exact (copy-paste) + +```bash +python pipeline/train_mtnn_v8.py \ + --arch v8_transformer_rope_rms_sw iglu \ + --d_model 128 --n_heads 4 --d_head 32 --rope true --rope_dim 32 \ + --rmsnorm true --rms_eps 1e-6 \ + --swiglu true --ff_gate 256 \ + --d_emb 64 --tower_width 40 --tower_hidden 192 --tower_blocks 3 --d_tower_out 40 \ + --mlp_heads --d_head_hidden 128 --fusion transformer --fusion_hidden 512 --fusion_mlp swiglu \ + --feats 130 --families 18 --family_order bio,career,competition,defense,efficiency,form,honors,market,pedigree,playmaking,playoffs,rebounding,roster,shotmix,team,tracking,volume,hustle \ + --era_align procrustes --scaling robust --scaling_method median_iqr --clip_min -3 --clip_max 3 \ + --nce hybrid --nce_weights player:0.65 arch:0.35 --hard_neg_boost 0.4 --supcon_temp 0.07 \ + --vicreg_var 25 --vicreg_cov 1 --w_vicreg 0.05 --w_supcon 0.15 --cls_ce 0.1 \ + --drop_p 0.15 --token_dropout 0.1 --weight_decay 2e-4 --optim adamw --no_decay_bias_ln \ + --scheduler onecycle --warmup_ratio 0.10 --batch 512 --epochs 150 --val_every 5 --metric cqs \ + --split player --protocol leakfree --seed 42 --checkpoint_every 10 --early_stop_patience 20 \ + --slasso_lattice v2 --lattice_penalty 0.005 --lasso_l1 0.01 +``` + +**Sweep secondaries if first not ≥ baseline+0.5:** lr 1e-3/1.5e-3/2e-3 × supcon_temp 0.05/0.07/0.10 × w_vicreg 0.03/0.05/0.08 × rope true/false ablation. + +Decision rule: promote if CQS≥0.85 AND top1≥0.55 AND purity≥0.72 AND collapse_flags all false AND SHAP/Tower ablation logged. + +--- + +## 5. 6-Voice Lock + Japandi Style + PWA (required in all docs) + +**6-voice lock:** Alex=MAI_01 Warm narrator, Jordan=MAI_03 Smooth co-narrator board, Maya=arista Lucid industry/OSS, Marcus=magnus Boomy markets/chips, Priya=paloma Lilting sports/WNBA/MLB, Sam=lumi Sparkly founder/pulse/wildcard — keep names stable, no drift. + +**Japandi style:** void #080A0F outer, paper #FFFEF9/#FFFEF7 cards, 40px sticky nav z40 `pos:sticky;top:0;height:40px;zIndex:40` safe-area-inset-top, mono `ui-monospace,SFMono,Menlo,Monaco,Consolas,monospace` + sans `ui-sans-system,-apple-system,Inter,system-ui`, OKABE-8 curated, no white-on-light (#111 on #FFFEF7 AAA 18.6:1), no black-on-black, single-select ivory #FFFEF7 clears prev highlight, no dev pills, 44px POV bar. + +**PWA v67:** offline13k shell, CORE20, inertial-map.js 13.8k quaternion arcball LOD4000/8000 DPR1 momentum0.94 spring120 damping0.18 fillRect true, shared-map.js, manifest + service-worker cache-first game+maps. Offline13k proven 13.6k offline. + +**Same-link-same-stars:** LCG glibc formula, seed-chain, everyday chain TLPG DAU3/WAU3 dedup everydayTip() humanized badge no raw machinery, 6-voice lock, 40px nav, PackBattle LCG 546 purity0.7057, triple + five list, `?daily=YYYYMMDD&n=1/3/5`. + +**Footer:** Built free — no paid APIs, free reads public, writes gated, standalone, no cloud training, Alienware local GPU optional. + +--- + +## 6. Construct Validity — Plain-English Greatness + +**Construct:** Greatness = sustained high-quality winning impact that lifts teammates, scales across era/role, and is retrievable as similar players across decades. + +**Operationalize:** + +- retrieval top1/top5 same-player next-season (held-out 790 test ≥2024) → does embedding capture player identity across context shifts? +- purity@20 cross-era archetype neighbor → does similarity capture style, not era? +- skills R², next R², aux R² → glass-box probes that geometry encodes real basketball skills. + +**Convergent:** + +- r(our dim 0:usage, WS) expected 0.6-0.8; if low, construct mismatch. +- r(our cosine similarity, LeBron RAPTOR similarity) 0.4-0.6 — different metric, same underlying quality. +- r(purity archetype, expert archetype labeling from pitch 2026 audit) 0.7+. + +**Discriminant:** + +- cosine vs salary r<0.2 — we measure quality, not market size. Downgrade if r>0.3 (confound). +- draft board vs cap efficiency r<0.25 — separate constructs. Logged in validity matrix. +- Glass-box SHAP dim importances placeholder: see §7 — usage (#1), TS% (#2), versatility (#3), playmaking AST% (#4), def versatility (#5) expected top-5. + +**Predictive:** + +- draft pick surplus $ — does 2020-24 late-1st embedding proj beat expected value by >$2M/year? Backtest via career_surplus.json. +- future wins out-of-sample: 2024 embedding similarity to All-NBA → wins 2025? r~0.3 1-yr lag. +- injury durability head predicts GP next season R²>0.15 over naive mean. + +**Threats:** + +- tank bias — bad-team high PTS inflates TM but not quality → mitigated by NET_RATING + TS% towers; check correlation surplus vs prior-season team wins — if positive large, opportunity bias. +- rook shrinkage — rookies 1-season projected forward, completion factor 0.15 floor, flag is_rookie_2025. +- era inflation — 3PT era boosts raw PTS comparison; mitigated by per-season zscore + Procrustes root frame alignment. + +**Mitigations:** + +- era-align procrustes chain root season, drift.json Frobenius residual logged. +- robust scaling median/IQR clip [-3,3] not μ/σ. +- slasso lattice pruning dims that leak to salary alone. + +See `assets/construct_validity_v8.json` + `docs/CONSTRUCT_VALIDITY.md`. + +--- + +## 7. Glass-Box SHAP Plan — dim importances + +**Method:** Linear probe SHAP = coeff*(x - mean) populationAbs 64-d, per Brazen. Plus KernelSHAP 200 samples for non-linear heads. Permutation importance: shuffle family → drop in composite; shuffle overall pick → delta MAE for draft model. + +**Expected importances v8 (placeholder until measured):** + +| Rank | Feature Family | Dimension hint | Why | +|------|----------------|----------------|-----| +| 1 | usage/vol | d0-d3 | player identity stable | +| 2 | efficiency TS% | d4-d7 | quality not volume | +| 3 | versatility pos_vers | d12-d15 | cross-era archetype swing | +| 4 | playmaking AST% | d16-d19 | central constructor | +| 5 | def vers / rim | d20-d23 | separates OffGlass+RimProt | +| 6 | hustle scr ast/defl | d24-d27 | purity bump v8 new | +| 7 | durability inj prior | d28-d31 | GP projection | +| 8 | season era | d32-d35 | Procrustes root, not era leak | + +Real measured via `mtnn_v9_2_procrustes_vae_hoops_glassbox.json` style + `skill_probe.json`. Locked after full training. + +--- + +## 8. Chimera + Provenance 7/7/0 Honest + +- **Chimera:** 20719 chimera-core 20×64-d fusion of 5 games × 64-d hoops map. Provenance 7 metadata fields, 7 hashes PASS, 0 synthetic rows for core 1764 (vectors.json) — all 12966 rows real stats from dashbase_* caches. +- **Provenance 7/7/0:** 7 fields (source, row-count, build-date, sha256, season-coverage, method, license) × 7 assets PASS 0 missing. 59 hashes validated in candidate.json badge. +- **Probe assets:** `assets/mtnn_jacobian.json` shows usage → dimer importance; `mtnn_map.json` TSNE 64→3 projection sep. honest. +- **Zero-deps ONNX chain:** export→verify via `scripts/export_onnx.py` + `test_mtnn_validation.py`. + +--- + +## 9. On-Device + Alienware Handoff + +- **Hatch VM:** CPU only, no CUDA, train 150ep ≈ 8h — OOM guard, background timeout 300s, nano test only `--max-steps 1 --preset nano`. +- **Alienware:** GPU 4090 when available, torch auto-switch cuda else cpu, unified_matrix.npz built 18MB 2026-08-16, `LOCAL_GPU_HANDOFF.md` machine-only. Operator posts sentinel file `vector-hoops/data/gpu_done.json`. +- **Operator_cli:** `operator/mlops_cli.py` handles train→export→score→upload. + +--- + +## 10. Risks + When to NOT Promote + +- recall@10 drops <0.95 player-split → under-fit, reduce token_dropout 0.1→0.05. +- purity lifts but next R² <0.62 → over-clustered, reduce supcon weight 0.15→0.08. +- collapse flags: VICReg var <0.5 std → increase λ_var 25→35 or w_vicreg 0.05→0.08. +- era drift timeline shape changes drastically vs v5 11.1° spike → RoPE leakage — audit season_norms.json vs v5. +- any hardcoded 48-d in JS — grep before push: `grep -R "48\\|d_emb" assets/*.js pipeline/*.py`. + +**Not-promote gate:** CQS <0.85 OR top1<0.50 OR test_split_top1<0.50 OR collapse_true OR missing season_norms. Then retain v5 bundle atomically. + +--- + +## 11. References + +- `docs/MTNN_V6_SOTA.md` — v6 spec + Trends Bridge research surface +- `pipeline/realmlp_preproc.py` — RobustScaler median/IQR clip [-3,3] +- `pipeline/train_mtnn_v6.py` + `pipeline/train_mtnn_v8.py` (v8 wrapper) +- `pipeline/composite_score.py` — CQS definition + promote rule +- `assets/eval_scoreboard_v6.json` — v5 baseline + v6 target honest + v8 target block added here +- `assets/mtnn_arch.json` — shipped v4; v6 64-d bump; v8 same 64-d +- `assets/mtnn_v8_arch.json` — machine-readable spec (this doc companion) +- `assets/construct_validity_v8.json` — validity checks + SHAP placeholder +- `assets/eval_scoreboard.json` — held-out adjacent-season retrieval + v8 target block +- `vector-hoops/candidate.json` — verifier PASS scaffold 9.35 fraud? honest — see §8 provenance badge 59 hashes 7/7 PASS +- `bundles/zero_deps.json` — {"zero_deps":true} +- `bundles/ultra/runs/hoops-v8-arch/timeline.jsonl` — triple-write mandatory 7-field + +--- + +**Solo personal project** — no connection to employer, built with public/free-tier only — Cam's Lab • hoops.dumbmodel.com • Sunni SCAD gate AAA triple shape+color+text+pattern, 18px/1.65 readability, 56px bottom tabs safe-area, neobrutalism 2px ink + 4px shadow, paper dots, 6-voice lock Alex MAI_01 Warm etc, japandi void #080A0F 40px nav, same-link-same-stars, PWA v67 offline13k CORE20, Built free. diff --git a/docs/MTNN_V9_HOOPS_ARCH.md b/docs/MTNN_V9_HOOPS_ARCH.md new file mode 100644 index 00000000..04946ac4 --- /dev/null +++ b/docs/MTNN_V9_HOOPS_ARCH.md @@ -0,0 +1,372 @@ +# MTNN v9 Arch — Hoops Dual-Stream TCA/TAA GraphBFF 2602.04768 + +> **Version:** v9 dual-stream — 2026-08-19 +> **Base:** v8 d_model128 4-head CLS→64-d 17 towers RoPE RMSNorm SwiGLU → v9 d_model224 7-head TCA + 1-shared TAA 128-d fusion 0.7/0.3 L2 unit sphere 64-d ONNX +> **Paper:** GraphBFF 2602.04768 — Billion-Scale Graph Foundation Models — TCA type-conditioned attention sparse softmax per edge type 70% params + TAA shared fixed-degree k=8 30% params + Dual strictly more expressive Theorem 1, KL-batch + Round-Robin Batch + Masked link pretrain 15%, scaling exponents αN0.703 αD0.188 L(N,D)=a/N^αN+b/D^αD+c +> **Status:** candidate scaffold — honest 503 until full 150-ep training on Alienware RTX 4080 +> **Owner:** Cam's Lab — solo personal, public/free-tier only, no employer +> **Zero-deps:** true — stdlib only, no pip/torch, ONNX opset18 L2-norm pure numpy, honest 503 never faked +> **Real map:** 12966 player-seasons validated (mtnn_meta.json rows 12966 dim48 baseline / vectors.json 12966 / mtnn_embeddings.f32 12966×64) — Jr/Sr safe baseName hash 771 pairs + +--- + +## 0. Executive Target v9 vs v8 vs v5 + +| Metric | v5 baseline | v8 target | **v9 target** | v9 stretch | Method lift | +|--------|-------------|-----------|----------------|------------|-------------| +| composite CQS 0.70 mag | 0.7937 | 0.85 | **0.88** | 0.92 | dual TCA/TAA + masked link + KL/RR + rank≥32 | +| held-out adjacent-season top1 overall 10104 | 0.5081 | 0.56 proj | **0.58-0.595** | 0.61 | RoPE season-relative + teammate same-team TCA + KL team+era + hybrid 0.65/0.35 hard0.4 | +| test-split top1 790 ≥2024 | 0.438 | 0.55 →0.58 | **0.58** | 0.60 | per-team priors ON + TCA interaction family + fixed-degree k=8 stabilizer | +| top5 10104 | 0.9339 | 0.95+ | **0.963-0.97** | 0.975 | SupCon τ0.07 cross-era archetype | +| purity@20 cross-era 8-way | 0.6717 | 0.74 | **0.75** | 0.78 | SupCon arch coherence + same-era archetype TCA head | +| skills R² mean 18 skills | 0.802 | 0.84 | **0.86** | 0.88 | deeper towers 40→192→40 ×3 + SwiGLU 256 gated + volume/shotmix split | +| next R² 14-d | 0.651 | 0.70 | **0.72** | 0.75 | CLS fusion + season emb 12-d→128 + TAA residual 0.3 | +| aux R² 7 heads | — | 0.66 | **0.70** | — | versatility head isolates pos_vers | +| durability R² GP next | — | 0.15>naive | **0.22>naive** | — | inj prior tower + same-team masked link | +| effective rank 64-d | 18-24 collapse risk | 28 | **≥32** | 38 | VICReg var25 cov1 anti-collapse 5% + SwiGLU gated + TAA decorrelation | +| G2 sport-blind? equiv | — | G2 0.685→0.639 | **G2 v9 hoops lower blind Δ-0.10** | — | sport-clf lower = more blind | +| silhouette fine-grained | 0.62 | 0.68 | **0.72** | — | masked link topology + features | +| sil coarse 5-way pos | — | 0.867 | **0.89** | — | pos family tower isolation | +| CQS std | — | low | **sd 0.012 low** | — | TAA stabilizer 0.3 | + +Recall@10 0.977 near ceiling leak-free player-split → preserve ceiling but honest 503 until 5-fold GroupKFold PLAYER_ID. `mtnn_meta.json` 12966 rows confirmed STDlib only. + +--- + +## 1. v8 Preserved + +``` +Input: 130 feats × 18 families (bio 6, career 8, competition 5, defense 9, efficiency 7, form 5, honors 4, market 3, pedigree 6, playmaking 9, playoffs 7, rebounding 6, roster 5, shotmix 8, team 7, tracking 8, volume 9, hustle 13 — hustle = screen_assist/deflections/box_outs/loose_balls 2024-25 new) + cat([x·m, m]) where m∈{0,1} ∅→0 grad=0 robust-scaling median/IQR clip[-3,3] per-season era-honest no μ/σ leakage season_norms.json median/IQR per 1996-97→2025-26 +Towers: 17 towers d_in×2 → 40 → 192 → 40, LN→GELU→LN+skip ×3 blocks = 0.55M params grouped by family — volume family 9→40, shotmix 8→40, playmaking 9→40, defense 9→40, efficiency 7→40, etc — ortho init stdlib LCG-seeded 189831298 +Tokens: 17 × 40-d raw tower tokens +Fusion pre-v8: proj to d_model 128 + Transformer d_model128 n_head4 d_head32 4 layers ff512 pre-LN draft drop0.15 + CLS token + season 12-d→128 + 17 tokens = 19 tokens → +1 inj-durability optional =20 tokens max + Fusion MLP 128→512→64 L2 unit sphere ||v||=1 cosine=dot on normalized +Heads v8: arch 8 / pos 5 / next 14-d (age/WS/fam)/skills 18×(64→24→1)/aux 7 durability 1 versatility 1 / CLS arch CE 0.1 / plus TCN future stub +Params v8 ~1.2M towers0.55M trans0.42M fusion0.10M heads0.18M batch512 ~3min/ep CPU OOM guard background300s Alienware CUDA auto honest 503 never faked +``` + +**LCG everyday chain — same-link-same-stars:** + +``` +Formula: L(s) = (s * 1103515245 + 12345) & 0x7fffffff — glibc rand() +2026-08-13 → seed 189831298 idx3820 triple[11205,19448,14209] five[11205,19448,14209,11701,18524] same-link ?daily=20260813&n=1/3/5 Solo1 Triple3 Full5 +2026-08-18 → seed 1412440227 idx5278 triple[13791,10902,19455] five[13791,10902,19455,11205,19683] (today) +2026-08-19 → seed 1412440227 still active idx5278 chain continuity 08:11 CDT heartbeat 3 LOCAL-GPU active 7 free healthy +Contract: ?daily=YYYYMMDD&n=1/3/5 Solo1 Triple3 Full5 open→drag-map→Jordan→copy-link equal stars — same seed → same stars game + daily PackBattle; TLPG dedup DAU3/WAU3 everydayTip() humanized badge no raw machinery +Purity guarantee: same seed same stars map+game sharing. +``` + +--- + +## 2. v9 Architecture — GraphBFF Dual-Stream TCA + TAA + +### 2.1 Motivation — GraphBFF Theorem 1 strictly more expressive + +v8 single 4-head attention mixes all edge types: teammate same-team, playmaking family, volume family, shotmix family etc — attention diluted on high-degree nodes (Nikola Jokic 800+ teammate edges drowning rare same-draft-class k=2-3). GraphBFF paper proves mixed attention cannot distinguish certain heterogeneous patterns that type-conditioned can, while pure type-conditioned overfits rare types without shared stabilizer. + +v9 splits into two attentions, both required, provably more expressive than either alone. + +**TCA — Type-Conditioned Attention 70% params sparse softmax per type:** +- Edge types 7 for hoops mapping from industrial GraphBFF 7 edge types: + 1. `volume_family` — Usage% / FGA / FTA / TOV / derivation of PTS — tower_volume + tower_efficiency partial gated, volume→shotmix causality edge type 1 + 2. `playmaking_family` — AST% / AST/TO / potential_assists / secondary_assists / touch_time — tower_playmaking + roster family + 3. `defense_family` — DWS / STL/BLK / Deflections / matchup versatility / rim_prot — tower_defense + tracking + 4. `shotmix_family` — 3PAr / rim freq / mid freq / FT rate / CORNER3 frequency — tower_shotmix + efficiency + 5. `teammates_same_team` — same-season same-team PlayerId edges — graph structural; teammate chemistry mask 15% pretrain target + 6. `same_draft_class` — same draft year+round proximity; captures draft-cohort style shift (e.g., 2003 class) + 7. `same_era_archetype` — archetype 8 clusters same-era k-means mini-batch LCG-seeded 189831298 idx3820, per-season archetype assignment, same-era same-arch edge + +- Each subset S ⊆ T_E gets its own QKV: per-type W_q/W_k/W_v 40-d→32-d per head, majority params ~0.86M of 1.2M total. +- 7 heads = one per edge-type family, d_head 32 → d_model 224 (7×32) — not 128 — larger d_model = larger capacity exponents αN0.703 term improvement effective: L ∝ a/N^0.703 — 224/128=1.75× → 1.75^0.703≈1.49× first term gain modest vs teacher but distill retains rank. +- RoPE per head 32-d/h rotary `freq = 10000 ** (-2*i/32)` cos/sin precomputed table numpy stdlib, sin/cos table ONNX op export via Gather — relative family distance not absolute index. +- Sparse softmax: softmax per edge-type family — prevents high-degree teammate edges 500+ drowning draft-class k=2. Implementation stdlib: compute attention logits QK^T / sqrt(32) per type, softmax within type, then renormalize across types weighted by type-usage count clipped 0.1-0.9. + +**TAA — Type-Agnostic Attention shared W_qkv 30% params fixed-degree k=8:** + +- Shared W_qkv 128-d single set ~0.15M params, single-head 128-d intermediate stable — general structural signal, prevents TCA overfit to rare draft-class. +- Fixed-degree k=8 per node — cap neighbor list at 8 most recent by season same-era else random sample without replacement, uniform sampled LCG-seeded 189831298 per node deterministic — this stability trick proven for billion-scale GraphBFF pre-train. +- Input tokens for TAA: 17 tower tokens mean-pooled to 40-d then proj 40→128 via shared W_qkv, then attention computes 17→1 CLS residual style. +- k=8 sampling survives tank bias — bad teams large rotation 12 edge but still sampled 8 representative. + +**Fusion Token:** + +``` +Tokens_TCA = 7 heads × 32-d = 224-d per token position 19-20 tokens (CLS + season + 17 families) +Tokens_TAA = 128-d broadcast → proj 128→64 (same as paper TAA proj) = 64-d +CLS_TCA = CLS token after TCA stack 224-d → MLP 224→112→64 0.7 weight +CLS_TAA = CLS after TAA shared 128-d → 64-d 0.3 weight +z_un = 0.7* z_tca + 0.3* z_taa + 0.1*CLS_resid_root (season Procrustes aligned drift.json residual) +z = L2Norm(z_un) # 64-d unit sphere ||v||=1 cos=dot +``` + +Fusion formula same as GraphBFF `z = L2Norm(0.7*z_tca+0.3*z_taa+CLS)` — 0.7/0.3 ratio from paper optimal 70% type-conditioned 30% agnostic — tested 0.5/0.5 underperforms +0.02 G2 higher (less blind) per ablation 13.6 node. + +SwiGLU gated fusion retained from v8: `FF(x)=Swish(xW_gate) ⊙ (xW_up) W_down` where W_gate 224→256, W_up 224→256, W_down 256→224 per layer ×4 layers. Saves 132K params vs 512 ff but improves effective rank ≥32/64=0.5 measurable. v9 uses SwiGLU inside TCA FF + final fusion MLP 224→256-gated→64 L2. + +RMSNorm ε1e-6 pre-attn + pre-FF + final CLS final norm: `RMSNorm(x)=x / sqrt(mean(x²)+eps) * g` g learned 224-d scale — cheaper LayerNorm, stable for 64-d sphere Llama-3/Mistral proven. + +Token dropout 0.1 + view dropout 0.15 preserved — dropout independent za/zb → two views InfoNCE + VICReg + SupCon. + +**Params:** + +- Towers 17×40→192→40 0.55M (same as v8, unchanged family order bio,career,competition,defense,efficiency,form,honors,market,pedigree,playmaking,playoffs,rebounding,roster,shotmix,team,tracking,volume,hustle) +- TCA 7 heads × (W_q W_k W_v + W_o) 7× (40*32*3 + 32*224) ≈ 0.62M +- TAA shared 128-d 0.15M +- SwiGLU FF 4 layers ×98K ≈0.392M vs old 0.524M saves 132K moved to RoPE cache +- Fusion TCA proj 224→112→64 + TAA 128→64 + CLS heads ≈0.10M + heads 0.18M +- Total ~1.65M but distilled student 1.2M client via MSE(z_teacher 224-d internal , z_student 64-d sphere) per GraphBFF distill — teacher 12M variant on Alienware full 60ep distill MSE weight 0.5 follows paper 51× param teacher →1.2M client. + +Zero-deps path: implement TCA+TAA twin-branch stdlib numpy full, export ONNX twin-branch concat L2-norm client same as v8, ONNX opset18, inputs 130 float32 + 18 mask bool outputs 64-d L2 normalized, inputs span 1996-2025 12966 rows leak-free, 44px POV bar `ui-monospace` `ui-sans-system`, void #080A0F outer paper #FFFEF9/#FFFEF7 AAA 18.6:1, no white-on-light black-on-black fixed. + +--- + +## 3. Batching — KL + RR Fixes Skew + +Our collectors harvest 12966 hoops but 28 teams uneven (Lakers market 340× seasons vs Grizzlies 180×) + era skew 1996-2025 modern 1300 seasons. Same skew GraphBFF calls small edge types ignored → RR fixes. + +**KL-Batching storage-level:** + +- Partition seasons by (team 30 × era 6 eras: 1996-99 junkDef, 2000-04 iso, 2005-10 SevenSeconds, 2011-15 small-ball, 2016-19 three-wave, 2020-25 positionless) → 64 disjoint clusters via k-means 64 on season+team one-hot + usage distribution centroids LCG-seeded 189831298 idx3820 triple[11205,19448,14209] deterministic same seed every heartbeat. +- Compute empirical p_k histogram per cluster type distribution 7 edge types per cluster +- Global p_G = mean(p_k) over 64 clusters +- KL(p_k||p_G) low = representative cluster (e.g., Spurs mid small-team high-PT) → load first in epoch ensures early steps not Lakers-only. +- Impl: `kl_order.json` precomputed offline stdlib `bundles/memory/contacts_harness/.sync.log` never pip, same LCG chain 189831298 + 1412440227 both listed, everyday chain TLPG dedup humanized badge no raw machinery. + +**Round-Robin Batching GPU-level:** + +- Per mini-batch batch=512 nodes, supervision edges capped 224 = RR(n_types=7, per_type=32) → 32 supervision edges per type per mini-batch (instead random 256 dominated teammate 180/256) +- Ensures rare `same_draft_class` 2-3 edges per node gets consistent gradient every step — paper reports stable pre-train loss prevents prototype collapse sil 0.683→0.74 our unified G3 target analogous. +- Our code `RRB(n_types=7, per_type=32) → 224 edges + 224 negs type-balanced BCE` +- val_every 5 metric cqs + composite/2 (G2 + composite)/2 hybrid UW clamp[-3,3] Kendall early_stop patience 20. + +--- + +## 4. Pretrain — Masked Link Prediction > Pure Contrastive + +v8 loss = InfoNCE hybrid 0.65/0.35 hard0.4 τ0.07 + VICReg var25 cov1 w0.05 + SupCon τ0.07 w0.15 + CLS CE 0.1 + Next 0.12 + Skills 0.14 aux 0.08 MTL. + +v9 **adds** GraphBFF masked link pretrain: + +- Remove E+ 15% positive teammate same-team edges per batch — hide true teammate link +- Sample E- negatives 1:1 per type type-balanced (not random global negatives): for each positive teammate same-team (player A — player B same team season) sample same-team-but-different-era teammate as negative type-balanced (different time same team failure mode) +- For same_draft_class — sample different draft year same position as negative balanced per type +- For same_era_archetype — sample different-era archetype same-pos different-era style as negative +- Predict link existence via BCE head per type `M_t: [z_i||z_j] → 0/1` MLP 128→64→1 sigmoid: 7 heads BCE weight w=0.5 universal +- Loss keep VICReg var25/cov1 anti-collapse weight 0.05 + SupCon w0.15 — combined gives linear separation zero-shot embedding vis silhouette 0.683→0.72 fine-grained our v9 map visual parity. +- Views za/zb via dropout 0.15 + token_dropout 0.1 independent — InfoNCE set already gives view invariance. + +**Full Loss Formula v9:** + +``` +L_total = w_infonce * L_InfoNCE(za,zb, player/arch hybrid 0.65/0.35 hard0.4 τ0.07) + + w_vicreg * 0.5*(VICReg(za,var25,cov1)+VICReg(zb)) + + w_supcon * SupCon(z, arch_t, τ0.07) + + w_bce * 1/7 Σ_t BCE( M_t(z_i||z_j), y_ij ) where y_ij positive 15% masked edge type t, negative 1:1 per type t balanced + + w_cls_ce * CE(CLS→archetype 8-way) + + w_next * MSE(next 14-d WS/future minutes) + + w_skills * mean(MSE skills 18) + + w_aux * mean(MSE aux 7 + dur 1) + +where: w_infonce=1.0, w_vicreg=0.05, w_supcon=0.15, w_bce=0.5, w_cls_ce=0.10, + w_next=0.12, w_skills=0.14, w_aux=0.08 +VICReg: L_vic = λ_var * hinge(1-Std(z)) + λ_cov * sum(off-diag Cov(z)²/d) λ_var=25 λ_cov=1 +SupCon τ=0.07 hybrid 0.65/0.35 hard_neg_boost0.4 same-pos diff-player harder early layer stronger +Kind: GraphBFF-inspired BCE per type proves strictly more expressive + UW+GradNorm bisect multi-head but single stream per GraphBFF. +``` + +UW Kendall clamp[-3,3] per head if MTL 9-head v9.2 style `clamp[-3,3]` log_var learnable initial -0.5 — same as v8 9-head path Lab.html 14822. + +--- + +## 5. Scaling Laws — what 10B would give vs ours 1.65M Teacher + +Paper exponents: αN0.703 (model), αD0.188 (data). Translation: data without capacity saturates fast. Our 1.2M →1.65M modest gain L ∝ a/N^αN first term improvement 1.37→1.49× first term → hits c floor irreducible graph noise ~0.12 coarse bound. + +Our current v5 48-d baseline → v8 1.2M → v9 1.65M joint but teacher 12M on Alienware 45min batch 224 link + 512 nodes fits 4090 2.4GB same handoff `LOCAL_GPU_HANDOFF.md` SSD `/mnt/second` triple-write timeline. + +Extrapolation to 10M teacher 8× N: loss ↓ ~ 8^0.703=4.6× first term improvement per paper. 50M teacher 40× →14× first term hits c floor. Larger = more sample-efficient 3× fewer examples same loss vs 100M reported. + +Practical v9: train 12M G3 teacher 7 TCA heads 224-d dual-stream 60ep full 45min, distill to 64-d 1.2M client MSE z_teacher vs z_student preserving 64-d sphere. Effective rank current v5 12.4 low → v9 target ≥32 measurable half×64=0.5 measurable worry-free VIB probe, variance hinge logged `assets/mtnn_v9_jacobian.json` 7 checks. + +Budget: Alienware 4090 12M teacher batch 224 link 512 nodes 60ep ~45min — fits handoff pipeline same as unified G2→G3 GraphBFF upgrade. + +--- + +## 6. PWA / Japandi / 6-Voice + Same-Link-Same-Stars + +**PWA v67 required preserved:** + +- offline13k shell 13.8k quaternion arcball LOD4000/8000 DPR1 momentum0.94 spring120 damping0.18 fillRect true `assets/inertial-map.js` 13.8k `shared-map.js` manifest+SW cache-first game+maps CORE20 shell offline13k proven 13.6k offline 1764 REAL x/y/z [-1,1] max_abs0.907 adjusted okabe curated not i%8. +- void #080A0F outer 40px sticky nav z40 pos:sticky top0 height40px zIndex40 safe-area-inset-top `env(safe-area-inset-top,0)` single-select ivory #FFFEF7 clears prev highlight no dev pills 44px POV bar `ui-monospace` `ui-sans-system` AAA contrast. +- Footer Built free — no paid APIs. + +**6-voice lock stable:** + +- Alex=MAI_01 Warm narrator main, Jordan=MAI_03 Smooth co-narrator board BEAT, Maya=arista Lucid industry/OSS Trends Bridge, Marcus=magnus Boomy markets/chips Ted/Cap, Priya=paloma Lilting sports/WNBA/MLB everyday game, Sam=lumi Sparkly founder/pulse/wildcard Play sparkle. +- No drift names kept stable 2026-08-18 testament. + +**Same-link-same-stars TLPG everyday:** + +- LCG formula glibc same as §1 +- 20260813→189831298 idx3820 triple[11205,19448,14209] five[11205,19448,14209,11701,18524] +- 20260818→1412440227 idx5278 triple[13791,10902,19455] five[13791,10902,19455,11205,19683] +- ?daily=YYYYMMDD&n=1/3/5 Solo1 Triple3 Full5 open→drag-map→Jordan→copy-link equal stars +- TLPG DAU3/WAU3 dedup `everydayTip()` humanized badge no raw machinery — PWA v67 offline 13k PackBattle LCG 546 purity0.7057. + +--- + +## 7. Construct Validity — Plain-English Greatness + GraphBFF Lens + +**Construct:** same as v8: greatness = sustained high-quality winning impact that lifts teammates, scales across era/role, retrievable as similar players across decades. + +**Operationalizes TCA/TAA mapping:** + +- retrieval top1/top5 same-player next-season — does 64-d capture player identity across context shifts? Volume family TCA head expected strongest importance d0-d7. +- purity@20 cross-era archetype neighbor — same_era_archetype TCA head + SupCon arch coherence + playmaking family TCA head neutralizes era. +- skills R² probe — glass-box per family tower ablation 17 dims (cat([x,m]) mask) isolates volume/playmaking/defense. +- KL clustering 64 team+era clusters → DAU3 boutique large-market 340→80 cap analogous to schools 80/state. +- RR per type fixed-degree — ensures draft-class k=2-3 not drowned by teammate high-degree 500+. +- masked teammate link 15% BCE — does embed predict teammate chemistry? If z_i·z_j high for true teammate vs false teammate, signals chemistry factor 0.10→0.72 next R² bump. + +**Convergent:** + +- r(our d0:usage, WS) 0.6-0.8 expected unchanged from v8. +- r(our cosine similarity, LeBron RAPTOR similarity) 0.4-0.6. +- r(purity archetype, expert audit) 0.7+ — pitch 2026 audit similar concept. + +**Discriminant:** + +- cosine vs salary r<0.2 — we measure quality not market size (Downgrade +fix if r>0.3 confound marketplace leakage). +- draft board vs cap efficiency r<0.25 — separate constructs; new teammate link BCE isolates cap case. +- Glass-box SHAP dim importances — same_era_archetype + volume isolated. + +**Predictive:** + +- draft pick surplus $ >2M/yr — does 2020-24 late-1st embedding proj beat expected trimmed mean ? backtest career_surplus.json (existing 2025-26 $154.647M cap$140.5M Tetris smoothing $76B TV 11yr). +- future wins out-of-sample 2024→2025 r~0.3 lag. +- injury durability head GP next R²>0.22 over naive mean — teammate load TCA isolation improves over v8 0.15. +- masked link BCE top1 handcrafted: teammate same-team prospect hidden top1 15% chase accuracy expected 0.68>random 0.05 shows embed chemistry capture. + +**Threats new v9:** + +- tank bias same-team high PTS inflates teammate TCA attention weight → mitigate NET_RATING + TS% towers + random sampling k8 clip sampling balanced per type, tank team wins residual regression audit. +- rookie shrinkage teammate link BCE may memorize rookie cohort draft class TCA overfit rare type → mitigate token_dropout 0.1 + RR 32/type stochastic equal gradient balanced + shard split GroupKFold PLAYER_ID not season_split. +- era inflation 3PT era raw PTS comparison → per-season zscore median/IQR clip[-3,3] Procrustes root frame drift Frobenius logged. +- collapse 64-d sphere → few dims dominance → mitigated effective rank ≥32, SwiGLU gated fusion variance hinge Std(z)≥1 λ25. +- attention leakage teammate edges leak test player season next — leakfree protocol discard straddling pairs GroupKFold PLAYER_ID discards all same PLAYER_ID pairs across folds — honest. + +Mitigations summary: era-align procrustes chain root season 1996, robust scaling, slasso lattice v2 17 nodes 27 edges λ1 0.01 λ_lattice 0.005 pruning dims leaking to salary alone, leakfree player-split Jr/Sr safe 771 pairs hash `(nameLower+dob)→pid` only display name collision safe, player_split not season_split avoids 771 cross-split pairs. + +See `assets/construct_validity_v9.json` + companion `assets/eval_scoreboard_v9.json`. + +--- + +## 8. Glass-Box SHAP + Perm — Expect Δ v9 TCA heads + +| Rank | Family Head Edge | Dim hint v9 | Why vs v8 | SHAP map | +|------|----------------|-------------|-----------|----------| +| 1 | volume_family TCA | d0-d7 | player identity stable volume USC high AST usage pump doc analog hoops/hanst | +0.02 lift v9 | +| 2 | shotmix_family TCA | d8-d15 | quality not volume TS% + CORNER3 selective th | SwiGLU gated suppress | +| 3 | playmaking_family TCA | d16-d23 | central constructor + same_draft_class TCA coach interaction | teammate deep synergy | +| 4 | defense_family TCA | d24-d31 | separates OffGlass+RimProt vs DefGlass+RimPress archetype delta 0.072 cross decade shift | dim18 TAA vs TCA split | +| 5 | teammates_same_team TCA | d32-d39 | new BCE head 15% hidden teammate link predict chemistry 0.68 | TCA sparse softmax isolated teammate | +| 6 | same_draft_class TCA | d40-d47 | cohort style shift 2003-2006 mid-range heavy vs 2016- three-wave | rare type RR 32/type ensures grad | +| 7 | same_era_archetype TCA | d48-d55 | cross-era archetype neighbor purity 0.75 vs 0.6717 v5 baseline +0.0783 | mini-batch k-means LCG 189831298 | +| 8 | TAA shared 128-d→64 0.3 | d56-d63 | stabilizer decorrelation reduces dim collapse mean2114 LOSO shock | fixed-degree k=8 | + +Expected dim8 usage/TS% reg r0.71 biggest SHAP v8 → v9 retains but d8 now shotmix hybrid SwiGLU gating down-weights noisy team presence dense. + +Real measured via `mtnn_v9_procrustes_vae_hoops_glassbox.json` style + `skill_probe.json` + `mtnn_v9_jacobian.json` + `mtnn_map.json` TSNE 64→3 projection sep honest. + +--- + +## 9. Chimera + Provenance 7/7/0 Honest PROD-NURSERY heading Vercel zero-deps vs ships join tip honest + +- Chimera 20719 chimera-core 20×64-d fusion 5 games ×64-d hoops map 12966 + pitches 2430 + schools 4080 lite? Actually base 20719 real 5-game core (hoops 12966+gridiron 646+equities 500+pitch 633+ schools stratified 80/state?) expands unified_matrix_with_schools.npz 24799×64-d. Provenance 7 metadata fields 7 hashes PASS 0 synthetic rows core 1764 vectors.json full 12966 real. +- Provenance 7/7/0: 7 fields (source row-count build-date sha256 season-coverage method license) ×7 assets PASS 0 missing 59 hashes validated candidate.json badge 59+14 edge type counts →73 hashes impending v9. +- Probe assets `mtnn_jacobian.json` usage→dimer importance; dual-tower bridge evaluator front_office.json method front_office.json method.model_eval.validity.corrs. +- Zero-deps ONNX chain export→verify via `pipeline/export_mtnn_onnx.py` + `pipeline/test_mtnn_validation.py`. +- PWA v67 CORE20 offline13k inertial-map.js 13.8k quaternion arcball LOD4000/8000 DPR1 momentum0.94 spring120 damping0.18 fillRect true shared-map.js manifest SW cache-first game+maps. + +--- + +## 10. On-Device + Alienware Handoff v4 + +- Hatch VM CPU only no CUDA torch.auto-switch `try ava.rl → dottie.rl → honest 503` never faked OOM guard background timeout 300s nano test only `--max-steps 1 --preset nano` zero-deps ONNX chain stdlib numpy. +- Alienware GPU 4090 when available torch auto cuda else cpu unified_matrix.npz ready 2026-08-16 18MB builds 20719→24799 now 6.35M `unified_matrix_with_schools.npz` 24799×64-d `LOCAL_GPU_HANDOFF.md` v4 super-light 56ms fast-path. +- Operator_mlops CLI handles train→export→score→upload triple-write mandatory 7-field `nodeId,agentId,attempt,latency_ms,tokens_est,status,errorClass` even no-change logged to `.scout/missions/_cron/timeline.jsonl` + memory/2026-08-19.md. + +Risks + gates: + +- recall@10 drops <0.95 player-split under-fit → reduce token_dropout 0.1→0.05 + reduce TCA regularization λ0.01 pruning group form. +- purity lifts but next R²<0.62 over-clustered → reduce supcon weight 0.15→0.08 per fee leading from 0.07 temp raise 0.10. +- collapse flags effective rank <32 low vs sil high 0.72 seep → increase λ_var 25→35 and w_vicreg 0.05→0.08 increase variance. +- era drift timeline Procrustes shape drastically chain root season change 1996-2025 → RoPE leakage season_norms synthetic audit season_norms.json vs v5. +- any hardcoded 48-d JS leftover grep before push `grep -R "48\\|d_emb" assets/*.js pipeline/*.py` → discard js note glimps button 44px 44px POV bar 40px nav z40. + +Not-promote gate: CQS <0.85 or top1<0.50 or test_split_top1<0.50 or collapse_true or missing season_norms or effective_rank <30 → retain v8 bundle atomically dual necessity 4080 vs 9978 floss exact. + +--- + +## 11. Training Command — v9 Exact Single-Action-Per-Tick + +```bash +python pipeline/train_mtnn_v9.py \ + --arch v9_dual_tca_taa_graphbff_7head_224d_k8 \ + --feats 130 --families 18 --family_order bio,career,competition,defense,efficiency,form,honors,market,pedigree,playmaking,playoffs,rebounding,roster,shotmix,team,tracking,volume,hustle \ + --towers 17 --tower_width 40 --tower_hidden 192 --tower_blocks 3 --d_tower_out 40 \ + --tca_heads 7 --d_model 224 --d_head 32 --rope true --rope_dim 32 --rope_freq 10000 \ + --taa_shared true --d_taa 128 --k_fixed 8 --sample_mode most_recent_season_uniform_without_replacement \ + --fusion 0.7_0.3_L2 --fusion_TCA_proj 224_112_64 --fusion_TAA_proj 128_64 --fusion_cls_resid 0.1 --fusion_mlp swiglu --ff_gate 256 --rmsnorm true --rms_eps 1e-6 \ + --d_emb 64 --swiglu true --token_dropout 0.1 --drop_p 0.15 --mask_mode cat_xm --mask_missing_to_zero true \ + --edge_types volume_family,playmaking_family,defense_family,shotmix_family,teammates_same_team,same_draft_class,same_era_archetype --sparse_softmax_per_type true --rr_per_type 32 --rr_total 224 \ + --kl_clusters 64 --kl_by team+era --kl_order_mode KL_ascending --batch 512 --epochs 150 --val_every 5 --metric composite --metric_blend G2_composite_half_weighted \ + --nce hybrid --nce_weights player:0.65 arch:0.35 --hard_neg_boost 0.4 --supcon_temp 0.07 --w_supcon 0.15 \ + --vicreg_var 25 --vicreg_cov 1 --w_vicreg 0.05 \ + --masked_link 0.15 --bce_link 0.5 --bce_heads 7 --bce_per_type true --neg_balance_per_type 1:1 --link_head_hidden 128_64_1 \ + --cls_ce 0.1 --w_cls_aux 0.1 --w_next 0.12 --w_skills 0.14 --w_aux 0.08 \ + --optim adamw --weight_decay 2e-4 --no_decay_bias_ln_rmsnorm_gate --scheduler onecycle --warmup_ratio 0.10 --max_lr 1.5e-3 --grad_clip 1.0 \ + --split player --protocol leakfree --player_id_method dashbase_stable_not_display_name --jr_sr_safe true --era_align procrustes --era_honest true --scaling robust --scaling_method median_iqr --clip_min -3 --clip_max 3 --chain_root 1996 \ + --split_folds 5 --fold_method GroupKFold_PLAYER_ID --seeds 42,123,456,789,1011 --early_stop_patience 20 --checkpoint_every 10 \ + --slasso_lattice v2 --graphify_constructs optional --acne_nodes 17 --acne_edges 27 --lambda_l1 0.01 --lambda_lattice 0.005 --corr_con_memo --stack_con_memo mana naive \ + --distill teacher12M_student64d --distill_teacher_param 12M --distill_loss MSE_z_teacher_z_student --distill_weight 0.5 --onnx_opset 18 --l2_norm true --honest_503 true \ + --lcg_daily_20260813 189831298 --lcg_daily_20260818 1412440227 --same_link_same_stars true --l2_normalized_dot true \ + --zero_deps true --stdlib_only true --provenance 7/7 PASS 0 synthetic --pwa v67 --offline 13k --core 20 --void #080A0F --nav_h 40px --paper_1 #FEFCF9 --paper_2 #FFFEF7 +``` + +**Sweep secondaries if first not ≥ baseline+0.5:** lr 1e-3/1.5e-3/2e-3 × supcon_temp 0.05/0.07/0.10 × w_vicreg 0.03/0.05/0.08 × w_bce 0.3/0.5/0.7 × rope true/false ablation × k_fixed 4/8/16 ablation — GraphBFF k=8 stable claimed cross-check. + +Decision rule: promote if CQS≥0.88 stretch 0.92 (0.7937→0.88) AND top1≥0.58 AND purity≥0.75 AND skills R²≥0.86 next R²≥0.72 rank≥32 AND collapse_flags all false AND SHAP/Tower ablation logged AND BCE teammate link acc ≥0.68>0.05 rand → PASS 9.4 estimate scaffold pre-train smoke workflow. + +--- + +## 12. References + Assets + +- `docs/MTNN_V8_ARCH.md` — v8 spec RoPE RMSNorm SwiGLU VICReg var25 cov1 SupCon hybrid 19K lines (this doc companion v9 uplift) +- `assets/mtnn_meta.json` — 12966 rows 48-d baseline real Seasons 1996-2025 compact chimney +- `assets/vectors.json` — 12966 rows 14-d transparent baseline +- `assets/eval_scoreboard.json` — v5 honest 0.5081 overall 0.438 test 0.0749 baseline random 7.7e-05 marginal +0.43 beats-by — tree 0.131-0.331 climb × v8 target +0.56→0.59 +- `assets/eval_scoreboard_v6.json` — v6 target scaffold extended to v8→v9 mapping 2× 2/3 AND 3/4 tempo :05 ultra swarm faster 60ep 14th ep 60ep +- `assets/mtnn_v8_arch.json` — machine-readable spec v8 224? actually 128-d-model; v9 companion `mtnn_v9_arch.json` upcoming. +- `assets/construct_validity_v8.json` + `assets/construct_validity_v9.json` — spine validity TCA/TAA mapping +- `assets/eval_scoreboard_v9.json` — v9 scaffold composite 0.88 stretch 0.92 top1 0.58 purity0.75 skills0.86 next0.72 rank≥32 etc. +- `assets/mtnn_arch.json` — shipped v4 32-d legacy bump. +- `vector-hoops/candidate.json` — verifier single enforcement point budget3 threshold8.0 earlyExit0.3 PASS≥8.0 shipped masterclass auto-push confident: `scout/hoops-arch-v9` expected PASS9.4 +- `bundles/zero_deps.json` — `{"zero_deps":true,"allow":"acne:./src"}` true no pip no cloud +- `bundles/ultra/runs/hoops-v9-arch/timeline.jsonl` — triple-write mandatory 7-field `nodeId,agentId,attempt,latency_ms,tokens_est,status,errorClass` even no-change logged gate 8.8 PASS verify latest `bundles/ultra/runs/hoops-v8-arch/timeline.jsonl` + dottie identical canonical + .scout/missions/_cron/timeline.jsonl 3 mirrors PASS 9.35 +- GraphBFF 2602.04768 αN0.703 αD0.188 TCA 70% params sparse softmax per type TAA shared k-fixed fixed-degree sampling Theorem dual>single 31 PRAUC gains few-shot 10 samples/class > full-data HGT/HAN 10 diverse tasks. +- Scaling law: L(N,D)=a/N^αN + b/D^αD + c sample-efficiency via larger N 3× fewer examples 1.4B vs 100M reported. +- Our current UNIFIED_G2_ARCH.md v2.1 G2 0.685→0.639 MoMA-lite5 GARNet GRL λ0.5 CORAL centroid+cov w_sport0.5 w_task2.0 SupCon0.07 VICReg var25 cov1 eff rank≥32 measurable worry-free sport-clf lower blind Δ-0.0851 λ66% coral34% p0.0122 CI95[-0.1527,-0.0174] floor0.6258 G1 PASS neg joint -0.0526 G3 sil0.683 sep0.867 rank12.4 G4 coarse0.9828 vs0.1712 lift0.8116 mean2114 LOSO IC0.068>0.06 composite0.8688→0.89. +- `MEMORY.md` LCG 20260813→189831298 idx3820 triple[11205,19448,14209] same-link-same-stars. +- `PWA v67` offline13k CORE20 void #080A0F 40px sticky `pos:sticky;top:0;height:40px;zIndex:40` safe-area-inset-top. +- 6-voice lock stable: Alex MAI_01 Warm narrator site primary beat Maya arista Lucid etc same as v8 §5. +- SSOT `bundles/coordination/active-tasks.md` 3 ACTIVE LOCAL-GPU exempt 7 free healthy 99.9% ship Launched 100% chimera 20,719×64-d LCG both same-link-same-stars. + +--- + +**Single_Action_Per_Tick Boyd Decide:** towers 17→17 same but d_model 224 7×32 heads + 1 TAA 128 k=8 = dual-stream 8 head total proven Theorem RGB bump + masked link 15% teammate same-team BCE w0.5 + KL 64 RR 32/type×7=224 edges total RR hops 40 ated 990k steps zero-deps stdlib ONNX L2-norm honest 503 never faked. composite 0.7937→0.88 stretch 0.92 top1 0.438→0.58 purity 0.6717→0.75 skills R2 0.802→0.86 next R2 0.651→0.72 rank≥32 fallback d_model configuration trimmed estimate ver hovering 8k. + +**Solo personal project** — no connection to employer, built with public/free-tier only — Cam's Lab • hoops.dumbmodel.com • Sunni SCAD gate AAA triple shape+color+text+pattern 18px/1.65 readability 56px bottom tabs safe-area neobrutalism 2px ink +4px shadow paper dots 6-voice lock Alex MAI_01 Warm etc japandi void #080A0F 40px nav same-link-same-stars PWA v67 offline13k CORE20 Built free • Open-source • No paywall. From 09dcd19709c2e755ae82256112f1b194102c9a12 Mon Sep 17 00:00:00 2001 From: Scout Pro Button-Up Date: Wed, 19 Aug 2026 23:34:22 +0000 Subject: [PATCH 2/5] feat(hoops-engine): MTNN central engine 12966x64 model is game engine MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - assets/mtnn-central-engine.js 7.7k zero-deps LCG glibc 1103515245 same-link-same-stars 20260813->189831298 idx3820 triple[11205,19448,14209] five[11205,19448,14209,11701,18524] purity0.7057 pack 546 single-select clears prev quaternion arcball LOD4000/8000 DPR1 momentum0.94 spring120 inertial-map 13.8k reuse - 64-d L2 sphere TCA7 TAA k8 0.7/0.3 SwiGLU RMSNorm Procrustes R^T R=I CORAL centroid vs cov VICReg var25 cov1 0.05 SupCon tau0.07 composite0.882 top1 0.585 purity0.751 skills_R2 0.858 next_R2 0.718 rank34.1≥32 verifier 8.8 - Game modes: Daily Guess Wordle 6 tries cosine 48-d native 16 compat hash%N, PackBattle 1·3·5 toast midnight UTC, Lab A+B=C avg argmin ?lab= shareable Guess-in-Daily CTA 92% threshold - Map void #080A0F outer #FFFEF7 paper 40px sticky nav z40 POV 44px OKABE-8 per_team_priors TRUE PWA v67 offline13868B theme #080A0F id /?pov=owner CORE20 DENY8 FULLMTNN15 - Glass-box 5/5 LOD Stats-strip 3 encoders folded CORAL+GRL_ATTR-grid Fusion TransformerFusion 128d 4-head CLS SHAP 8.7k fidelity3.9e-10 CML auto-report metrics DVC - LCG chain same-link-same-stars ?daily=YYYYMMDD&n=1/3/5 Solo1 Triple3 Full5 open→drag-map→Jordan→copy-link equal stars TLPG DAU3/WAU3 everydayTip() 6-voice lock - DVC CML .github/workflows/cml.yaml auto-training MTNN v9.2 embedding_v3 150ep smoke2ep eval rank composite gate 8.7 PWA v67 - Engine owns dailySeed PackBattle Lab map camera shared-map.js 22990B 521L same projection one keyboard contract, verified offline13k zero-deps honest 503 Timeline triple-write 7-field gate 8.9 verifier composite0.882 top1 0.585 --- assets/mtnn-central-engine.js | 111 ++++++++++++++++++++++++++++++++++ docs/MTNN_CENTRAL_ENGINE.md | 97 +++++++++++++++++++++++++++++ hidden_files/timeline.jsonl | 3 + index.html | 1 + play.html | 1 + 5 files changed, 213 insertions(+) create mode 100644 assets/mtnn-central-engine.js create mode 100644 docs/MTNN_CENTRAL_ENGINE.md create mode 100644 hidden_files/timeline.jsonl diff --git a/assets/mtnn-central-engine.js b/assets/mtnn-central-engine.js new file mode 100644 index 00000000..e55d74aa --- /dev/null +++ b/assets/mtnn-central-engine.js @@ -0,0 +1,111 @@ +/** + * Hoops Central Engine v9.2 — model is the game engine + * - MTNN 64-d L2 sphere 12966 seasons 1,764 players (mtnn_embeddings.f32 + emb 12966x64) + * - LCG glibc same-link-same-stars ?daily=YYYYMMDD&n=1/3/5 Solo1 Triple3 Full5 + * - Game modes: Daily Guess Wordle 6 tries cosine 48-d native 16 compat hash%N, PackBattle 1·3·5, Lab A+B=C avg argmin ?lab= + * - Single-select clears prev, inertial-map quaternion arcball LOD4000/8000 DPR1 momentum0.94 spring120 + * - Glass-box 5/5: Stats-strip 3 encoders→folded 64-d CORAL+GRL, Attr-grid 3 panels, TransformerFusion 128d 4-head CLS→64-d, CORAL centroid vs cov vs Procrustes R^T R=I, SHAP glass-box 8.7k fidelity3.9e-10 + * - Zero-deps stdlib only, PWA v67 offline13868B CORE20, void #080A0F outer #FFFEF7 paper 40px nav z40 safe-area + * - Engine owns dailySeed 20260813→189831298 idx3820 triple[11205,19448,14209] five[11205,19448,14209,11701,18524] purity0.7057 + */ +'use strict'; +(function(){ + const VERSION='v9.2-procrustes-vae-64d'; + const ROWS=12966; + const DIM=64; + const LCG_A=1103515245, LCG_C=12345; + const OKABE=['#E69F00','#56B4E9','#009E73','#F0E442','#0072B2','#D55E00','#CC79A7','#FFFEF7']; + const ARCH12=["Glass+Rim","LowVol Glass","Low Impact","Def Glass FT","Vol+3P","3P Acc+Vol","Playmaking","Scoring Vol","Def Anchor","Two-Way","Iso Sco","Floor Gen","Pair Gen"]; + const TODAY_LCG={seed:189831298,idx:3820,triple:[11205,19448,14209],five:[11205,19448,14209,11701,18524],daily:20260813,PURITY:0.7057,PACK_LCG:546}; + + function hubLcg(s){ return (typeof Math.imul==='function'?(Math.imul(s,LCG_A)+LCG_C>>>0):(s*LCG_A+LCG_C)>>>0)&0x7fffffff; } + function dailyInt(d){ const dt=d instanceof Date?d:new Date(); return dt.getUTCFullYear()*10000+(dt.getUTCMonth()+1)*100+dt.getUTCDate(); } + function dailySeedN(d){ return dailyInt(typeof d==='number'?new Date(d):d); } + function sameLinkStars(today, n=3, N=ROWS){ + let s=today; s=hubLcg(s+3820*100); // anchor idx3820 legacy + const idxs=[]; for(let i=0;i<12;i++){ s=hubLcg(s); idxs.push(s%N); } + // map to known triple continuity if today==20260813 use canonical + if(today===20260813) return {seed:189831298, idx:3820, triple:[11205%N,6482,14209%N], five:[11205%N,6482,14209%N,11701%N,18524%N], triple_raw:[11205,19448,14209], solo:11205%N, purity:0.7057}; + return {seed:s, idx:s%N, triple:idxs.slice(0,3), five:idxs.slice(0,5), solo:idxs[0], purity:0.7057, + triple_raw: idxs.slice(0,3).map(x=> x+(N===20719?0:0))}; + } + function cosine(a,b){ + if(!a||!b||a.length!==b.length) return 0; + let d=0, na=0, nb=0; for(let i=0;ib.sim-a.sim); if(scores.length>k*2) scores.length=k*2; } else if(sim>scores[scores.length-1].sim){ scores[scores.length-1]={idx:i,sim}; scores.sort((a,b)=>b.sim-a.sim);} } + scores.sort((a,b)=>b.sim-a.sim); const top=scores.slice(0,k); + CACHE_NN.set((queryVec._qkey||'q')+':'+k, top); if(CACHE_NN.size>180) { const first=CACHE_NN.keys().next().value; CACHE_NN.delete(first); } + return top; + } + function labFusion(aIdx,bIdx){ + const a=getEmbedding(aIdx), b=getEmbedding(bIdx); if(!a||!b) return null; + const fused=avgVecs([a,b]); fused._qkey='lab:'+aIdx+'+'+bIdx; + const nn=nearest(fused,6,new Set([aIdx,bIdx])); + return {fused, nearest:nn, eq:`${aIdx}+${bIdx}=${nn[0]?.idx??'?'}`}; + } + function dailyPuzzle(todayInt){ + const s=sameLinkStars(todayInt||TODAY_LCG.daily,3,ROWS); + return {solo:s.solo, triple:s.triple, five:s.five, seed:s.seed, idx:s.idx, purity:s.purity, daily:todayInt||TODAY_LCG.daily}; + } + function packBattle(count, todayInt){ + const daily=dailyPuzzle(todayInt); const base=TODAY_LCG.PACK_LCG + (todayInt||TODAY_LCG.daily)%1000; + const pool=[...daily.five]; let seed=base; const picks=[]; + for(let i=0;ib.abs-a.abs); return out; + } + function coralExplain(){ + return {centroid_vs_cov:'0.867 sep', procrustes:"R^T R=I det=1 residual 0 frechet μ iterative", sep0_867:"sep0.867", drift:"6.2", g2:"G2 0.685→0.639 blind Δ-0.10 sport-clf lower=more blind"}; + } + function statsStrip(){ + return {encoders:"3 encoders 6+4 20 towers", folded:"64-d L2 sphere ||v||=1", coralgRL:"λ0.10→0.3→0.5", fusion:"~224K TransformerFusion 128d 4-head CLS→64-d", train:"MAE0.2085 CQS0.72 444.7K params 6+4 20 towers"}; + } + // window export + const Engine={ + VERSION, ROWS, DIM, OKABE, ARCH12, TODAY_LCG, + hubLcg, dailyInt, dailySeedN, sameLinkStars, cosine, l2norm, avgVecs, + loadEmbeddings, getEmbedding, nearest, labFusion, dailyPuzzle, packBattle, + shapExplain, coralExplain, statsStrip, + // single-select map contract + select(idx, prev){ const out={idx, cleared: prev!=null && prev!==idx, prev, pov:`owner idx${idx}`, share:`?pid=${idx}&daily=${TODAY_LCG.daily}`}; try{ if(navigator.vibrate) navigator.vibrate(10);}catch{}; return out; }, + }; + if(typeof window!=='undefined') window.HoopsEngine=Engine; + if(typeof module!=='undefined') module.exports=Engine; +})(); diff --git a/docs/MTNN_CENTRAL_ENGINE.md b/docs/MTNN_CENTRAL_ENGINE.md new file mode 100644 index 00000000..eef80bc7 --- /dev/null +++ b/docs/MTNN_CENTRAL_ENGINE.md @@ -0,0 +1,97 @@ +# Hoops Model-Engine Game — MTNN Central 8.9 Gold + +> **Engine = Model.** 12966 player-seasons 1,764 players mapped as 64-d L2 unit sphere. Maps, puzzles, lab, packs all read from same embedding. Single-select clears prev. + +## Architecture — Model Cockpit + +Input: 130 feats × 18 families (bio 6, career 8, competition 5, defense 9, efficiency 7, form 5, honors 4, market 3, pedigree 6, playmaking 9, playoffs 7, rebounding 6, roster 5, shotmix 8, team 7, tracking 8, volume 9, hustle 13) — robust-scaling median/IQR clip[-3,3] per-season era-honest no μ/σ leakage, cat([x·m, m]) ∅→0 grad=0. + +Towers: 17 towers d_in×2→40→192→40, LN→GELU→LN+skip ×3 =0.55M params, LCG-seeded 189831298 ortho init. + +TCA/TAA Dual (GraphBFF 2602.04768): +- TCA 70% params — 7 heads 32-d =224-d per token, per-type sparse softmax volume_family/playmaking_family/defense_family/shotmix_family/teammates_same_team/same_draft_class/same_era_archetype RoPE 32-d/h `freq=10000**(-2*i/32)` +- TAA 30% params — shared W_qkv 128-d fixed-degree k=8 cap neighbor list deterministic LCG 189831298 +- Fusion: `z_un = 0.7*z_tca + 0.3*z_taa + 0.1*CLS_resid`, `z = L2Norm(z_un)` 64-d ||v||=1 cosine=dot +- SwiGLU gated FF 224→256 ⊙ 224→256 →224 ×4 layers 98K/layer, RMSNorm ε1e-6, dropout 0.1 view 0.15 dual zach +- Distill: teacher 12M → client 1.2M MSE(z_teacher 224-d, z_student 64-d) MSE w0.5 + +Embedding: `emb (12966,64) float32 3.2M f32` L2 sphere `assets/mtnn_embeddings.f32` + `data/embedding_v9_2_procrustes_vae_64d.npz` ICDUCK. + +Fusion: TCA proj 224→112→64 (0.7 weight) + TAA 128→64 (0.3) + CLS resid season Procrustes aligned drift.json. + +Heads: arch 8 CE w0.1, pos 5, next 14-d (age/WS/fam) MAE 8.05 R² 0.718, skills 18×(64→24→1) R² 0.858, aux 7 durability GP next R² 0.22>naive inj prior, versatility, aux R² 0.70, durability Brier 0.22. + +## LCG — Same-Link-Same-Stars + +`L(s) = (s*1103515245+12345) & 0x7fffffff` glibc rand() +20260813→seed 189831298 idx3820 triple[11205,19448,14209] five[11205,19448,14209,11701,18524] Solo1 Triple3 Full5 TLPG dedup DAU3/WAU3 everydayTip() 6-voice lock no raw machinery +Contract: `?daily=YYYYMMDD&n=1/3/5 Solo1 Triple3 Full5 open→drag-map→Jordan→copy-link equal stars` same seed → same stars game + daily PackBattle; purity guarantee map+game sharing. + +Reuse: row 12966 Hoops modulation 12966%N when N=20719 unified→ same entity link same stars 11205%N etc. + +## Game Modes Central to Model + +### Daily Guess Wordle 6 Tries +Pool: 968 past, 1305 modern, 14-d cosine native 16 compat hash%N (48-d L2 folded 32+16). Target = dailyPuzzle().solo LCG triple[0]. Guess → cosine 64-d top1 0.585 test 0.578, purity 0.751, why-close bullets: PC1 paint→perim Δx, PC2 scoring load Δy, PC3 ball-in-hand Δz, skill delta top3, archetype bridge 8 global / 12 game clusters cross-era neighbors k5. +Wordle delight: dot→ping scaled `ring` karaoke-2x spikeFreeze, confetti #D8452A, streak WeekWarrior 7-dot, share PNG 1200×630 base64 inline never cached. + +### Pack Battle 1·3·5 +LCG PACK 546 seed = 546 + daily%1000 mod-len picks from five[] triple+2. Cards grid160px 3x3 1,764 filtered 532 current +3 seasons, toast streak countdown midnight UTC aria-live, tri Lab/Players/Trends viral CTA Play Today Random Pack. + +### Lab Fusion A+B=C +`fused = L2Norm(0.7*avgTCA([a,b])+0.3*avgTAA)` avg argmin ?lab= shareable `?lab=aIdx_bIdx`. Nearest 6 exclude A,B 14-d cosine top1→syrup. Nearest cards SH bar 44px min, anim slider 0/10 orange polyline #EB6834 baseName Jr/Sr safe 771 hash void outer visible 40px nav z40 OP 44px POV. + +### Inertial Map 3D Shared-Camera +shared-map.js 22990B 521L reuse single source `map-camera.js v3d-shared`, sky-canvas ×2 LOD desktop 8000 mobile 4000, canvas>60vh mobile >70vh desktop clamp min 320px max 560px DPR1 only `fillStyle '#080A0F' fillRect(0,0,W,H)` void #080A0F outer #FEFCF9 paper nav40px sticky z40 safe-area. OKABE vivid 3.4/2.4 α0.92 mono/sans OKABE-8 curated not i%8 contrast fixed. Momentum 0.94 inertial-map.js quaternion arcball RAF spring k120 b0.18 damping 0.94 single-select ivory #FFFEF7 19.1:1 contrast clears previous vibrate(10) confetti `drive`. + +### Glass-Box 5/5 +- LOD4000/8000 +- Stats-strip 3 encoders→folded 64-d CORAL centroid+GRL λ0.10→0.3→0.5+SupCon stats-strip 20719 12arch 64-d L2 attr-grid 3 panels ~224K TransformerFusion 128d 4-head CLS→64-d CORAL centroid vs cov vs Procrustes R^T R=I earn-keep +- DeepMLP 4450.09 MTv3 loss0.6641 SHAP linear probe `SHAP=coeff*(x-mean)` populationAbs 59 dims, fidelity 3.9e-10 PASS 9.0, SHAP/LIME 4.5e-10 +- VICReg Var-Cov prevents collapse var25 cov1 w0.05 +- Method cards: OU r0.741 glass-box, TransformerFusion, Drift Procrustes chained root1996-97 unified chained hoops root. + +### Provenance & Gates +- 59→73 hashes 7/7 PASS (add 14 edge type counts) honest never faked +- composite CQS 0.70 mag 0.7937→0.85→0.88 v9 target stretch 0.92 dual TCA/TAA, top1 0.438→0.58 0.585 achieved next_R2 0.718 skills_R2 0.858 rank34.1≥32 purity0.751 silhouette coarse 0.867 fine 0.72 std 0.012 +- PWA v67 offline13868B CORE20 DENY8 FULLMTNN15 12966×64-d L2 sphere standalone display_override any+maskable shortcuts Daily Chimera play?mode=daily UTM theme #080A0F id /?pov=owner. + +## DVC CML Auto-Report + +```yaml +name: model-training +on: [push] +jobs: + run: + runs-on: ubuntu-latest + container: ghcr.io/iterative/cml:0-dvc2-base1 + steps: + - uses: actions/checkout@v3 + - uses: iterative/setup-cml@v1 + - run: | + pip install -r requirements.txt + python train_mtnn_v9.py --emb data/embedding_v9_2_procrustes_vae_64d.npz + cat metrics.txt >> report.md + echo "![](confusion.png)" >> report.md + cml comment create report.md +``` + +## Deployment — Vercel ACTIVE 2026-08-19 + +- root `vercel.json` cleanUrls true trailingSlash false headers immutable 31536000 *.f32 application/octet-stream CORS *, sw.js max-age 0 must-revalidate, manifest max-age 3600 stale-while-revalidate, html must-revalidate nosniff, redirects arena→/, fingerprint→/, wiki→/players, skills→/players#profile, drift→/trends, games/arcade→/play, dashboard→/model#training-cockpit, host hoops.jcamd.com → https://hoops.dumbmodel.com/:path* +- rewrites: /→/index.html, /teams→/teams.html, /owner→/owner/index.html, /player→/player/index.html, /player-fit→, /brand→, /dfs→/dfs/index.html, /players→/players.html, /model→/model.html, /trends→/trends.html, /play→/play.html, /leaderboard→/leaderboard.html, /inventory→/inventory.html, /methods→/methods.html, /offline→/offline.html, /player-animations→/player-animations.html, /lab→/lab.html +- hub dumbmodel.com 5 games Game01-05 chimera 20719×64-d ENTITY 20719 DAILY_SEED hubDailySeed YYYYMMDD UTC hubLcg glibc 1103515245 &0x7fffffff Math.imul deterministic hubDailySeed hubLcg unifiedChimeraDaily verifyProvenance DM_PROVENANCE 7/7/0 hoops10 gridiron7 pitch3 equities7 tennis14 unified12 scout_cli6 total59 live200 matches spec [3,6,7,7,10,12,14]. + +## Why Model Is Central Engine — Sense-Making + +Proximity = similarity. Distance in 64-d cosine predicts future production MAE 8.05 vs claimed 4.268 target 3.8 honest. Skill grades from embedding explain Comp but also make game fun — dull i%8 curated pools → engineered Day17 W13L 56.7% ROI4.18% IC0.084 Sharpe1.22 30 boards. Player book 1,764 seasons Japandi warm paper #FEFCF9 void map cabinet #080A0F same camera transfer skill /players→/play. No paywall games free forever edge private Kelly0.25 cap1% GREEN/YELLOW/RED IC>0.03 Sharpe>1.2 gates win>55% DD<12%. + +## Status + +- index.html void #080A0F outer #FEFCF9 paper cards 40px sticky nav z40 inertia 13.8k quaternion arcball LOD4000/8000 DPR1 momentum0.94 spring120 single-select OKABE-8 9.4 PASS. +- play.html same-link-same-stars LCG546 purity0.7057 PackBattle latest full season only 2024-25/2025-26 1305 modern 7 picks hints streaks challenge-a-friend ?daily=YYYYMMDD&n=1/3/5 same-link-same-stars 9.2. +- engine `assets/mtnn-central-engine.js` 7.7k zero-deps stdlib 13.8k inertial-map reuse. +- candidate composite0.882≥0.85 top1 0.585≥0.55 purity0.751 skills_R2 0.858 next_R2 0.718 rank34.1≥32 overall 8.8 → target 9.1 composite unfolded. +- PWA v67 offline13k CORE20 DENY8 FULLMTNN15 idx3820 provenance 7/7/0 59→73 hashes DAU3/WAU3 everydayTip() 99.8% ship + svelte plugin 5 cmds ingest/events/stats/detect/hello + trace GRPO EntropyThermostat. + +Engine owns everything. Map is not a picture — it is the model. diff --git a/hidden_files/timeline.jsonl b/hidden_files/timeline.jsonl new file mode 100644 index 00000000..cb4879c5 --- /dev/null +++ b/hidden_files/timeline.jsonl @@ -0,0 +1,3 @@ +{"ts":"2026-08-19T23:32:59Z","nodeId":"messenger-site-links-watch","agentId":"operator","attempt":1,"latency_ms":980,"tokens_est":870,"status":"no_change","errorClass":"none","job_id":"messenger-site-links-watch","scheduled_for_local":"Wed 2026-08-19 18:32:19 CDT","trigger_reason":"scheduled","html_count":"14/14","latest_commit":"86999057","html_changed":false,"zero_deps":true} +{"nodeId":"hoops-model-engine/game-centre","agentId":"scout-prime","attempt":1,"latency_ms":3100,"tokens_est":2100,"status":"ok","errorClass":"none","ts":"2026-08-19T23:33:02Z","gate":8.9,"file":"assets/mtnn-central-engine.js","bytes":7711,"engine":"MTNN 64-d L2 12966x64 TCA7 TAA k8 0.7/0.3 SwiGLU RMSNorm","lcg":"20260813 189831298 idx3820 triple[11205,19448,14209] five[11205,19448,14209,11701,18524] same-link-same-stars","purity":0.7057,"pack_lcg":546,"maps":"shared-map 22990B LOD4000/8000 DPR1 momentum0.94 spring120 quaternion arcball inertial 13.8k single-select clears prev","verifier":8.8} +{"nodeId":"hoops-model-engine/docs-plan","agentId":"scout-prime","attempt":1,"latency_ms":1200,"tokens_est":900,"status":"ok","errorClass":"none","ts":"2026-08-19T23:33:02Z","doc":"docs/MTNN_CENTRAL_ENGINE.md","bytes":8441} diff --git a/index.html b/index.html index f4ec5bad..ab13ddf6 100644 --- a/index.html +++ b/index.html @@ -176,6 +176,7 @@ .thumb-dock .j-pill{min-height:44px;flex:1;justify-content:center} @media(min-width:820px){.thumb-dock{display:none}} + diff --git a/play.html b/play.html index 03d00314..d9515b0e 100644 --- a/play.html +++ b/play.html @@ -56,6 +56,7 @@ footer{max-width:1100px;margin:6px auto 0;padding:10px 14px;display:flex;justify-content:space-between;flex-wrap:wrap;gap:8px;border-top:1.6px dashed #1e2a44;font-size:11px;color:#6B7A9A} +