{
 "named_head_indices": {
  "1": "favorite",
  "4": "reply",
  "5": "quote",
  "6": "repost",
  "11": "dwell",
  "13": "video_quality_view"
 },
 "checkpoint": "oss-phoenix-artifacts (mini Phoenix, frozen)",
 "corpus": "sports_corpus.npz (~537K real posts, 6h window, Sports topic)",
 "head_distributions": {
  "0": {
   "name": "unnamed_0",
   "mean": 0.0,
   "median": 0.0,
   "p90": 0.0,
   "p99": 1e-05
  },
  "1": {
   "name": "favorite",
   "mean": 0.07728,
   "median": 0.06494,
   "p90": 0.16797,
   "p99": 0.25977
  },
  "2": {
   "name": "unnamed_2",
   "mean": 0.00016,
   "median": 7e-05,
   "p90": 0.0003,
   "p99": 0.0015
  },
  "3": {
   "name": "unnamed_3",
   "mean": 0.0,
   "median": 0.0,
   "p90": 0.0,
   "p99": 1e-05
  },
  "4": {
   "name": "reply",
   "mean": 0.00022,
   "median": 0.00013,
   "p90": 0.00052,
   "p99": 0.00129
  },
  "5": {
   "name": "quote",
   "mean": 0.0001,
   "median": 6e-05,
   "p90": 0.00025,
   "p99": 0.00059
  },
  "6": {
   "name": "repost",
   "mean": 0.00823,
   "median": 0.00476,
   "p90": 0.01965,
   "p99": 0.04883
  },
  "7": {
   "name": "unnamed_7",
   "mean": 0.0,
   "median": 0.0,
   "p90": 0.0,
   "p99": 1e-05
  },
  "8": {
   "name": "unnamed_8",
   "mean": 0.0,
   "median": 0.0,
   "p90": 0.0,
   "p99": 0.0
  },
  "9": {
   "name": "unnamed_9",
   "mean": 0.0,
   "median": 0.0,
   "p90": 0.0,
   "p99": 1e-05
  },
  "10": {
   "name": "unnamed_10",
   "mean": 0.0,
   "median": 0.0,
   "p90": 0.0,
   "p99": 1e-05
  },
  "11": {
   "name": "dwell",
   "mean": 0.2759,
   "median": 0.29688,
   "p90": 0.45703,
   "p99": 0.57816
  },
  "12": {
   "name": "unnamed_12",
   "mean": 0.62472,
   "median": 0.58594,
   "p90": 0.91406,
   "p99": 0.98438
  },
  "13": {
   "name": "video_quality_view",
   "mean": 0.07108,
   "median": 0.06689,
   "p90": 0.13672,
   "p99": 0.21191
  },
  "14": {
   "name": "unnamed_14",
   "mean": 0.00033,
   "median": 0.0002,
   "p90": 0.00073,
   "p99": 0.00198
  },
  "15": {
   "name": "unnamed_15",
   "mean": 0.00014,
   "median": 9e-05,
   "p90": 0.00025,
   "p99": 0.00067
  },
  "16": {
   "name": "unnamed_16",
   "mean": 0.0,
   "median": 0.0,
   "p90": 1e-05,
   "p99": 2e-05
  },
  "17": {
   "name": "unnamed_17",
   "mean": 2e-05,
   "median": 1e-05,
   "p90": 3e-05,
   "p99": 0.0001
  },
  "18": {
   "name": "unnamed_18",
   "mean": 0.0,
   "median": 0.0,
   "p90": 0.0,
   "p99": 0.0
  }
 },
 "n_scored": 50000,
 "implied_ratios_vs_favorite": {
  "favorite": 1.0,
  "reply": 0.0029,
  "quote": 0.0013,
  "repost": 0.1065,
  "dwell": 3.5703,
  "video_quality_view": 0.9199
 },
 "head_rank_correlations": {
  "favorite~reply": 0.629,
  "favorite~quote": 0.582,
  "favorite~repost": 0.424,
  "favorite~dwell": 0.835,
  "favorite~video_quality_view": 0.648,
  "reply~quote": 0.859,
  "reply~repost": 0.513,
  "reply~dwell": 0.564,
  "reply~video_quality_view": 0.391,
  "quote~repost": 0.664,
  "quote~dwell": 0.574,
  "quote~video_quality_view": 0.426,
  "repost~dwell": 0.503,
  "repost~video_quality_view": 0.49,
  "dwell~video_quality_view": 0.844
 },
 "age_curves": {
  "favorite": {
   "ages_h": [
    0.25,
    0.5,
    1.0,
    2.0,
    4.0,
    8.0,
    16.0,
    24.0,
    36.0,
    48.0,
    64.0,
    72.0,
    80.0,
    96.0,
    120.0
   ],
   "mean_prob": [
    0.07347,
    0.07347,
    0.07486,
    0.08373,
    0.08087,
    0.06289,
    0.07022,
    0.0744,
    0.0624,
    0.07633,
    0.08712,
    0.08354,
    0.09344,
    0.09344,
    0.09344
   ],
   "relative_to_first": [
    1.0,
    1.0,
    1.0188,
    1.1396,
    1.1006,
    0.8559,
    0.9558,
    1.0125,
    0.8493,
    1.0389,
    1.1858,
    1.137,
    1.2718,
    1.2718,
    1.2718
   ]
  },
  "reply": {
   "ages_h": [
    0.25,
    0.5,
    1.0,
    2.0,
    4.0,
    8.0,
    16.0,
    24.0,
    36.0,
    48.0,
    64.0,
    72.0,
    80.0,
    96.0,
    120.0
   ],
   "mean_prob": [
    0.00035,
    0.00035,
    0.00035,
    0.00033,
    0.00032,
    0.00023,
    0.00033,
    0.00023,
    0.00016,
    0.00018,
    0.00011,
    0.00012,
    0.00013,
    0.00013,
    0.00013
   ],
   "relative_to_first": [
    1.0,
    1.0,
    1.0008,
    0.9459,
    0.8914,
    0.6406,
    0.921,
    0.6439,
    0.4473,
    0.4953,
    0.3112,
    0.3516,
    0.364,
    0.364,
    0.364
   ]
  },
  "quote": {
   "ages_h": [
    0.25,
    0.5,
    1.0,
    2.0,
    4.0,
    8.0,
    16.0,
    24.0,
    36.0,
    48.0,
    64.0,
    72.0,
    80.0,
    96.0,
    120.0
   ],
   "mean_prob": [
    0.00015,
    0.00015,
    0.00015,
    0.00017,
    0.00015,
    9e-05,
    0.00012,
    0.00012,
    6e-05,
    8e-05,
    6e-05,
    6e-05,
    6e-05,
    6e-05,
    6e-05
   ],
   "relative_to_first": [
    1.0,
    1.0,
    1.0192,
    1.1333,
    1.0154,
    0.6288,
    0.8214,
    0.784,
    0.4304,
    0.548,
    0.3811,
    0.3858,
    0.4085,
    0.4085,
    0.4085
   ]
  },
  "repost": {
   "ages_h": [
    0.25,
    0.5,
    1.0,
    2.0,
    4.0,
    8.0,
    16.0,
    24.0,
    36.0,
    48.0,
    64.0,
    72.0,
    80.0,
    96.0,
    120.0
   ],
   "mean_prob": [
    0.00708,
    0.00708,
    0.00757,
    0.01075,
    0.01105,
    0.00833,
    0.00877,
    0.00865,
    0.00534,
    0.00551,
    0.00413,
    0.00366,
    0.00397,
    0.00397,
    0.00397
   ],
   "relative_to_first": [
    1.0,
    1.0,
    1.069,
    1.5186,
    1.5615,
    1.1769,
    1.2397,
    1.222,
    0.7544,
    0.7785,
    0.5837,
    0.5175,
    0.5611,
    0.5611,
    0.5611
   ]
  },
  "dwell": {
   "ages_h": [
    0.25,
    0.5,
    1.0,
    2.0,
    4.0,
    8.0,
    16.0,
    24.0,
    36.0,
    48.0,
    64.0,
    72.0,
    80.0,
    96.0,
    120.0
   ],
   "mean_prob": [
    0.28426,
    0.28426,
    0.28126,
    0.25895,
    0.26292,
    0.22527,
    0.25378,
    0.30239,
    0.24692,
    0.30316,
    0.30352,
    0.30935,
    0.31387,
    0.31387,
    0.31387
   ],
   "relative_to_first": [
    1.0,
    1.0,
    0.9894,
    0.9109,
    0.9249,
    0.7925,
    0.8927,
    1.0638,
    0.8686,
    1.0665,
    1.0678,
    1.0883,
    1.1042,
    1.1042,
    1.1042
   ]
  },
  "video_quality_view": {
   "ages_h": [
    0.25,
    0.5,
    1.0,
    2.0,
    4.0,
    8.0,
    16.0,
    24.0,
    36.0,
    48.0,
    64.0,
    72.0,
    80.0,
    96.0,
    120.0
   ],
   "mean_prob": [
    0.0572,
    0.0572,
    0.05808,
    0.06425,
    0.0644,
    0.05128,
    0.05967,
    0.09473,
    0.06501,
    0.0775,
    0.05121,
    0.06449,
    0.04876,
    0.04876,
    0.04876
   ],
   "relative_to_first": [
    1.0,
    1.0,
    1.0154,
    1.1234,
    1.1259,
    0.8966,
    1.0433,
    1.6562,
    1.1366,
    1.355,
    0.8954,
    1.1275,
    0.8525,
    0.8525,
    0.8525
   ]
  }
 },
 "n_fetched_live": 1311,
 "n_joined": 1311,
 "backtest": {
  "spearman_Pfav_vs_actual_likes": -0.314,
  "spearman_Preply_vs_actual_replies": -0.128,
  "spearman_Pfav_vs_actual_replies": -0.222,
  "likes_by_Pfav_decile": [
   187.5,
   98.0,
   41.0,
   31.0,
   27.0,
   21.0,
   22.0,
   25.0,
   25.0,
   18.0
  ]
 },
 "feature_effects": {
  "has_photo": {
   "n_with": 459,
   "median_likes_with": 44.0,
   "median_likes_without": 28.0,
   "median_replies_with": 1.0,
   "median_replies_without": 1.0,
   "mean_Pfav_with": 0.07744,
   "mean_Pfav_without": 0.07836
  },
  "has_video": {
   "n_with": 285,
   "median_likes_with": 62.0,
   "median_likes_without": 28.0,
   "median_replies_with": 1.0,
   "median_replies_without": 1.0,
   "mean_Pfav_with": 0.07198,
   "mean_Pfav_without": 0.07972
  },
  "has_link": {
   "n_with": 74,
   "median_likes_with": 15.5,
   "median_likes_without": 35.0,
   "median_replies_with": 0.0,
   "median_replies_without": 1.0,
   "mean_Pfav_with": 0.08802,
   "mean_Pfav_without": 0.07744
  },
  "hashtags_3plus": {
   "n_with": 26,
   "median_likes_with": 20.0,
   "median_likes_without": 34.0,
   "median_replies_with": 0.0,
   "median_replies_without": 1.0,
   "mean_Pfav_with": 0.09214,
   "mean_Pfav_without": 0.07775
  },
  "is_reply": null,
  "verified_author": {
   "n_with": 615,
   "median_likes_with": 56.0,
   "median_likes_without": 22.5,
   "median_replies_with": 2.0,
   "median_replies_without": 1.0,
   "mean_Pfav_with": 0.06432,
   "mean_Pfav_without": 0.09015
  },
  "question_mark": {
   "n_with": 92,
   "median_likes_with": 43.5,
   "median_likes_without": 32.0,
   "median_replies_with": 4.0,
   "median_replies_without": 1.0,
   "mean_Pfav_with": 0.07499,
   "mean_Pfav_without": 0.07827
  },
  "slop_pattern": null,
  "bait_pattern": null
 },
 "within_author_backtest": {
  "n_authors": 74,
  "n_posts": 877,
  "note": "audience size held constant: rank correlation of model P(action) vs actual counts WITHIN each prolific author's posts (>=6 live posts each)",
  "fav_spearman_median": -0.003,
  "fav_spearman_mean": -0.015,
  "fav_frac_positive": 0.5,
  "fav_iqr": [
   -0.187,
   0.157
  ],
  "reply_spearman_median": 0.049,
  "reply_frac_positive": 0.565
 },
 "replication": {
  "date": "2026-07-31",
  "what": "adversarial replication: all headline findings rerun under a second viewer context (empty engagement history) before publication",
  "within_author_cold": {
   "median_spearman": 0.008,
   "frac_positive": 0.51,
   "n_authors": 74,
   "verdict": "NULL REPLICATES - the model's inability to pick an author's winners holds across both viewer contexts"
  },
  "age_curves_cold_relative": {
   "favorite": [
    1.0,
    1.0,
    0.9967,
    0.9467,
    0.8329,
    0.7086,
    0.34,
    0.4051,
    0.4626,
    0.5512,
    0.9618,
    0.7185,
    1.0161,
    1.0161,
    1.0161
   ],
   "reply": [
    1.0,
    1.0,
    0.9993,
    1.0029,
    0.9234,
    0.598,
    0.9364,
    0.5952,
    0.5346,
    0.5866,
    0.4312,
    0.4643,
    0.4776,
    0.4776,
    0.4776
   ],
   "quote": [
    1.0,
    1.0,
    1.0448,
    1.3721,
    1.1648,
    0.6142,
    0.8958,
    0.6743,
    0.3971,
    0.5184,
    0.3299,
    0.3326,
    0.3358,
    0.3358,
    0.3358
   ],
   "repost": [
    1.0,
    1.0,
    1.1188,
    1.8276,
    1.9874,
    1.5002,
    1.7467,
    1.4993,
    1.0909,
    1.0492,
    0.8376,
    0.7048,
    0.7078,
    0.7078,
    0.7078
   ],
   "dwell": [
    1.0,
    1.0,
    0.9882,
    0.9079,
    0.9929,
    1.0089,
    1.0232,
    1.1409,
    1.0469,
    1.0203,
    1.1998,
    1.1066,
    1.1739,
    1.1739,
    1.1739
   ],
   "video_quality_view": [
    1.0,
    1.0,
    1.0016,
    1.0188,
    1.0477,
    0.9382,
    0.9557,
    1.4356,
    1.1754,
    1.1627,
    1.1371,
    1.1431,
    1.0535,
    1.0535,
    1.0535
   ]
  },
  "corrections": [
   "The 'favorites RISE with age (1.27x at 80h)' wrinkle was viewer-specific: the cold context shows flat (~1.0). Robust claim: favorites show NO decay. Reply and repost decay directions replicate in both contexts (cold reply 0.43-0.48 at 64-80h; cold repost 0.71-0.84)."
  ]
 },
 "checkpoint_forensics": {
  "note": "Measured directly from the shipped weight files (model_params.npz / embedding_tables.npz), not from documentation. These settle claims no code-reader could settle.",
  "readme_contradiction_resolved": {
   "root_README_claims": "256-dim embeddings, 4 attention heads, 2 transformer layers",
   "phoenix_README_claims": "mini version (128-dim, 4-layer transformer)",
   "measured_truth": "128-dim, 4 decoder layers (transformer/decoder_layer_0..3), 4 heads, key_size 32",
   "verdict": "phoenix/README.md is correct; the root README is wrong on BOTH dimension and depth",
   "verified_quote_root_README_line_32": "A pre-trained mini Phoenix model (256-dim embeddings, 4 attention heads, 2 transformer layers) is now packaged as a ~3 GB archive",
   "verified_phoenix_README_table": "Embedding dimension 128 | Transformer layers 4 | Attention heads 4",
   "why_it_propagated": "Syndicated coverage repeats 256-dim/2-layer because it faithfully copies X's own root README. The error is upstream, in the release's documentation of the artifact it ships."
  },
  "parameter_census": {
   "ranker_transformer_params": 807104,
   "retrieval_transformer_params": 833921,
   "embedding_params_each_model": 384000000,
   "grand_total_params": 769641025,
   "embeddings_share_pct": 99.79,
   "transformer_share_pct": 0.213,
   "interpretation": "The checkpoint is ~770M parameters of which 99.79% are ID-hash embedding tables (3 x 1M x 128 per model). The transformer that does the actual reasoning is 1.6M parameters total. This is the mechanistic explanation for the null backtest: the model is overwhelmingly a memorized lookup over IDs it saw in training, not a content evaluator."
  },
  "structural_receipts": {
   "post_age_embedding_table_rows": 82,
   "post_age_arithmetic": "82 = POST_AGE_MAX_MINUTES(4800) / granularity(60) + 2 special buckets - the 80-hour ceiling is physically baked into the shipped tensor shape, not just asserted in source",
   "action_projection_rows": 19,
   "continuous_unembeddings_cols": 8,
   "mlp_hidden_width": 176,
   "mlp_note": "176 = SwiGLU-adjusted widening (128 x 2.0 x 2/3, rounded to a multiple of 8) - consistent with widening_factor=2.0 in the published config"
  },
  "corpus_discrepancy": {
   "readme_claims": "~537K sports posts",
   "measured": 84564,
   "note": "the shipped sports_corpus.npz contains 84,564 post ids - 6.4x fewer than documented"
  },
  "action_count_contradiction_resolved": {
   "verified_2026_07_31_against_local_clone": true,
   "root_README": "states NO number - says the model 'predicts probabilities for many actions' (line 334) and lists 'each action type (like, reply, repost, click, etc.)' (line 201)",
   "phoenix_README": "19 (architecture table, line 267)",
   "third_party_reports": "public write-ups have variously reported 12, 13+, 14 and 15+ action types",
   "measured_truth": 19,
   "evidence": "phoenix_model/action_projection tensor is shape (19, 128) in the shipped ranker weights",
   "note": "An earlier draft of this entry claimed the root README says 14. We checked the file: it states no number. Corrected before publication - the same discipline this site asks of everyone else."
  }
 },
 "prior_art": {
  "checked": "2026-07-31 sweep of GitHub (repo has issues/discussions/PRs all DISABLED - zero public technical surface), Hacker News Jan+May threads, technical Substacks, forks, arXiv",
  "finding": "Every substantive public analysis of this release is STATIC CODE READING. No published work loads and runs the released checkpoint.",
  "nearest_prior_art": {
   "project": "hjosugi/xalgo",
   "url": "https://github.com/hjosugi/xalgo",
   "what": "computes Spearman/NDCG against a real For-You ordering snapshot, but with a heuristic proxy (engagement/views) - explicitly does NOT use the Phoenix checkpoint or its weights",
   "difference": "we run the released model itself; they approximate it with a hand-built proxy"
  },
  "notable_code_readings": [
   {
    "who": "Truth Tide",
    "url": "https://algo.truthtide.tv/technical.html",
    "verdict": "best static analysis found; correctly states production config did not ship"
   },
   {
    "who": "swyx (HN)",
    "url": "https://news.ycombinator.com/item?id=46688271",
    "verdict": "identified the weighted scorer and candidate-isolation masking from source"
   }
  ]
 },
 "calibration_check": {
  "note": "Cross-head probability scales from the released checkpoint are NOT calibrated to reality. Measured against ground truth in the same 1,311 posts.",
  "model_mean_Pfav_over_mean_Preply": 352,
  "actual_likes_over_replies_same_posts": 35.4,
  "actual_total_likes": 458436,
  "actual_total_replies": 12939,
  "miscalibration_factor": 9.9,
  "external_anchors_like_to_reply": {
   "X_heavy_ranker_design_statement_2023": 27,
   "Socialinsider_70M_posts": 15,
   "Metricool_1.12M_posts": 12.9,
   "Slaughter_et_al_arxiv_2502.13322_per_view": 7.0
  },
  "verdict": "The model's per-head RANKINGS may be usable; its CROSS-HEAD RATIOS are not. Anyone converting these probabilities into 'a reply is worth N likes' claims is publishing a training artifact. We made this error in our first draft and corrected it.",
  "retracted_claim": "Our initial /lab draft reported 'a predicted reply is ~345x rarer than a like, a quote ~770x rarer' as a fact about X. It is a property of the checkpoint's calibration, not of the platform. Retracted 2026-07-31, same day, before launch."
 },
 "corpus_temporal_span": {
  "posts_live": 1311,
  "range": "2026-05-14T20:36 -> 2026-05-15T20:19",
  "span_hours": 23.7,
  "implication": "The shipped corpus is a ~24h window (README says 6h). No before/after test of any policy change is possible with this data - stated so nobody mistakes our sample for a longitudinal one."
 },
 "age_curve_mechanism": {
  "correction": "The 80-hour flatline is NOT learned behavior - it is feature saturation, and we can prove it from the weights: post_age_embedding_table has exactly 82 rows (4800 min / 60 min + 2). Past bucket 80 every post shares one embedding, so the curve MUST be flat. Presenting it as a learned preference would have been wrong.",
  "what_remains_learned": "The SHAPE inside the 0-80h window (replies decaying, reposts peaking at 2-4h, favorites flat) is learned behavior and replicates across two viewer contexts."
 },
 "external_corroboration": {
  "links": {
   "finding": "Two peer-reviewed 2026 studies measure a large real-world link penalty that the open ranking code contains no rule for.",
   "studies": [
    {
     "cite": "Galeazzi et al., NDSS 2026 (arXiv 2410.17390)",
     "n": "40M+ tweets, 9M+ users",
     "result": "link posts 4.67-7.45x lower median visibility (views/followers); gradient by link type: none 0.246 > other-social 0.150 > twitter-internal 0.097 > news 0.033; p<2.2e-16"
    },
    {
     "cite": "Efstratiou et al., ICWSM 2026 (arXiv 2512.06129)",
     "n": "376 participants, ~205k exposures, paired counterfactual feeds",
     "result": "external links = strongest negative predictor (p<0.001); VIDEO EXEMPT; follower count non-significant (p=0.14) - which dissolves the account-size confound"
    },
    {
     "cite": "Buffer, 18.8M posts / 71k accounts",
     "n": "Aug 2024-Aug 2025",
     "result": "non-Premium link posts fell to 0% median engagement after Mar 2025"
    },
    {
     "cite": "Yu et al., CSCW 2025 (arXiv 2509.08128)",
     "n": "642,108 tweets, 2018 data",
     "result": "CONTRADICTS: URLs associated with HIGHER engagement in 2018 - suggesting a post-acquisition policy change rather than an intrinsic property of links"
    }
   ],
   "official_denials": [
    {
     "who": "Nikita Bier (X product)",
     "when": "Jul 2026",
     "claim": "'Links were never deboosted'"
    },
    {
     "who": "Elon Musk",
     "when": "2026-07-29",
     "claim": "'We haven't for over a year'"
    }
   ],
   "our_position": "No link term exists in the open ranking code - that part of our myth verdict stands and is code-checkable. But the behavioral evidence for a real penalty is now strong and peer-reviewed, and it is exactly the 'learned or hidden in withheld layers' case our verdict allowed for. We update the verdict rather than defend it."
  },
  "questions": {
   "our_finding": "posts containing a question earned ~4x median replies (1,311 posts)",
   "status": "No published study estimates this effect on X - confirmed by academic sweep. Genuinely unmeasured.",
   "caveat": "Neubrander et al. (arXiv 2601.16040, RCT n=2,282) found curiosity-promoting design raised question-asking but DECREASED liking. Questions may buy replies at a cost in likes; our data does not test the trade-off."
  },
  "decay": {
   "prior_work": "Pfeffer et al. (arXiv 2302.09654): median tweet impression half-life ~80 minutes, peak at 72 seconds. Measures IMPRESSION VOLUME, a different quantity from our predicted per-impression propensity. No 2024-2026 replication exists (API restrictions killed the line of work)."
  }
 },
 "transparency_regression": {
  "claim": "The 2026 release is a regression on the one dimension the 2023 release was genuinely praised for.",
  "2023": "twitter/the-algorithm published the actual engagement weights (fav 0.5, reply 13.5, report -369...). Arvind Narayanan called it the first time a major platform published its engagement calculation formula.",
  "2026": "xai-org/x-algorithm publishes the formula's SHAPE (sum of weight_i x P(action_i)) and withholds every coefficient - crate::params is not in the repo, verified by us at commit 0bfc279.",
  "compounding": "xAI's own headline claim is that they 'eliminated every single hand-engineered feature and most heuristics'. Removing interpretable features while withholding the weights moves the system further from outside inspection, not closer.",
  "governance": "The 2023 repo invited issues and PRs. The 2026 repo has issues, PRs, discussions and wiki ALL DISABLED, three squashed commits authored by 'CI agent', and a license change from AGPL-3.0 to Apache-2.0. It is a publication, not a project."
 },
 "composition_test": {
  "test": "impression-invariant engagement composition",
  "why": "view counts are unavailable from the zero-API path (syndication payload exposes favorite_count + conversation_count only, no views, no follower count). Composition cancels impressions algebraically instead.",
  "random_sample": {
   "n_posts": 1065,
   "spearman_model_vs_actual_reply_share": 0.046,
   "bootstrap_95ci": [
    -0.022,
    0.1
   ],
   "significant": false,
   "actual_reply_share_median": 0.0204,
   "model_reply_share_median": 0.00262
  },
  "prolific_authors": {
   "n_posts": 798,
   "spearman_model_vs_actual_reply_share": 0.173,
   "bootstrap_95ci": [
    0.09,
    0.227
   ],
   "significant": true,
   "actual_reply_share_median": 0.0044,
   "model_reply_share_median": 0.00336,
   "within_author": {
    "n_authors": 64,
    "median_spearman": 0.084,
    "frac_positive": 0.609
   }
  },
  "baseline_comparison": {
   "note": "same task, same posts: can a one-line heuristic (does the text contain '?') predict reply share better than the released model?",
   "model_spearman": 0.046,
   "question_mark_heuristic_spearman": 0.16,
   "n": 1065
  },
  "memorization_gradient": {
   "question": "Does the model's skill track how much of an author it could have memorized?",
   "method": "same impression-invariant composition test, posts split by how many times that author appears in the shipped corpus",
   "buckets": [
    {
     "author_posts_in_corpus": "1",
     "n": 417,
     "spearman": -0.033
    },
    {
     "author_posts_in_corpus": "2-4",
     "n": 384,
     "spearman": 0.052
    },
    {
     "author_posts_in_corpus": "5-10",
     "n": 171,
     "spearman": 0.168
    },
    {
     "author_posts_in_corpus": "11+",
     "n": 93,
     "spearman": 0.129
    }
   ],
   "reading": "Skill is absent for authors seen once and rises to ~0.13-0.17 for authors seen repeatedly. Not perfectly monotone (the top bucket dips, and is the smallest), but the direction is clear and it matches the parameter census: this is a memorization engine, and it predicts where it has memorized.",
   "caveat": "Corpus frequency is a proxy for training exposure, not a measurement of it. The corpus is a ~24h sports slice, so 'appears often here' and 'was trained on heavily' are correlated but not identical."
  },
  "view_counts_unavailable": {
   "checked": "2026-07-31",
   "finding": "The zero-API syndication endpoint returns favorite_count and conversation_count only - no view/impression count, and no follower count either (user object carries id, name, screen_name, verified flags, avatar).",
   "consequence": "A literal rate-per-view backtest is not possible without authenticated access. The composition test is the impression-invariant substitute and is arguably the better test anyway: it cancels reach algebraically rather than adjusting for it statistically."
  }
 }
}