#!/usr/bin/env python3 """Simulated A/B test for the AI music filter. SYNTHETIC DATA THROUGHOUT. Usage: python3 experiment.py # run with config/experiment.yaml python3 experiment.py --seed 7 # different random draw, same design Reads the pre-registered design (arms, segments, effects, thresholds) from config, simulates 30-day retention per user, and writes a readout that applies the decision rules written down BEFORE the run in 2026-08-18-ab-test-design.md. Everything here is a demonstration of experiment discipline, not a measurement of real users. The word SYNTHETIC appears in every output on purpose. """ import argparse import json import math from datetime import date from pathlib import Path import numpy as np import yaml Z = {0.05: 1.959964, 0.01: 2.575829} def two_proportion_test(success_a, n_a, success_b, n_b): """Two-proportion z-test. Returns (lift, z, p_value, ci_low, ci_high). lift is B minus A in percentage points, with a 95% CI. """ p_a, p_b = success_a / n_a, success_b / n_b pooled = (success_a + success_b) / (n_a + n_b) se_pooled = math.sqrt(pooled * (1 - pooled) * (1 / n_a + 1 / n_b)) z = (p_b - p_a) / se_pooled if se_pooled else 0.0 # two-sided p from the normal cdf p_value = 2 * (1 - 0.5 * (1 + math.erf(abs(z) / math.sqrt(2)))) se_unpooled = math.sqrt(p_a * (1 - p_a) / n_a + p_b * (1 - p_b) / n_b) ci_low = (p_b - p_a) - Z[0.05] * se_unpooled ci_high = (p_b - p_a) + Z[0.05] * se_unpooled return (p_b - p_a) * 100, z, p_value, ci_low * 100, ci_high * 100 def required_n_per_arm(baseline, mde_pp, alpha=0.05, power=0.80): """Sample size per arm for a two-proportion test.""" z_a, z_b = Z[alpha], 0.841621 # z for power .80 p1, p2 = baseline, baseline + mde_pp / 100 return math.ceil( ((z_a + z_b) ** 2 * (p1 * (1 - p1) + p2 * (1 - p2))) / ((p2 - p1) ** 2) ) def simulate(cfg, seed): rng = np.random.default_rng(seed) n = cfg["users_per_arm"] baseline = cfg["baseline_retention"] segments = cfg["segments"] seg_names = list(segments) seg_shares = [segments[s]["share"] for s in seg_names] results = {} for arm, arm_cfg in cfg["arms"].items(): seg_draw = rng.choice(len(seg_names), size=n, p=seg_shares) active = np.zeros(n, dtype=bool) retention_p = np.full(n, baseline) for i, seg_index in enumerate(seg_draw): seg = seg_names[seg_index] adoption = arm_cfg["filter_active_rate"].get(seg, 0.0) if rng.random() < adoption: active[i] = True retention_p[i] += segments[seg]["effect_pp"] / 100 retained = rng.random(n) < retention_p # guardrail: streams per week, slightly reduced for active filters in # the AI-positive segment (their catalogue got thinner) streams = rng.normal(cfg["baseline_streams_per_week"], 8, n) for i, seg_index in enumerate(seg_draw): if active[i] and seg_names[seg_index] == "ai_positive": streams[i] *= 0.96 results[arm] = { "n": n, "retained": int(retained.sum()), "retention": float(retained.mean()), "filter_active": int(active.sum()), "filter_active_rate": float(active.mean()), "streams_per_week": float(streams.mean()), "by_segment_active": { seg: int(sum(1 for i in range(n) if active[i] and seg_names[seg_draw[i]] == seg)) for seg in seg_names }, } return results def verdict(lift_pp, p_value, threshold_pp): if p_value < 0.05 and lift_pp >= threshold_pp: return "SHIP: clears the pre-registered threshold" if p_value < 0.05: return "REAL BUT BELOW THRESHOLD: do not ship as-is; effect exists but is too diluted" return "NOT SIGNIFICANT: the hypothesis is unproven at this sample size" def write_readout(cfg, results, seed, out_path): control = results["control"] lines = [ "# A/B Test Readout: AI Music Filter", "", f"**SYNTHETIC DATA.** Simulated per the pre-registered design " f"(2026-08-18-ab-test-design.md), seed {seed}, run {date.today().isoformat()}. " "Every effect below was produced by the assumption model in " "config/experiment.yaml, not by real users.", "", f"Baseline retention {cfg['baseline_retention']:.0%}, " f"{cfg['users_per_arm']:,} users per arm " f"(power requires {required_n_per_arm(cfg['baseline_retention'], cfg['ship_threshold_pp']):,}).", "", "| Arm | Retention | Lift (pp) | 95% CI | p | Filter active | Streams/wk |", "|---|---|---|---|---|---|---|", f"| Control | {control['retention']:.2%} | — | — | — | " f"{control['filter_active_rate']:.1%} | {control['streams_per_week']:.1f} |", ] stats = {} for arm in [a for a in results if a != "control"]: r = results[arm] lift, z, p, lo, hi = two_proportion_test( control["retained"], control["n"], r["retained"], r["n"] ) stats[arm] = (lift, p) lines.append( f"| {arm.upper()} | {r['retention']:.2%} | {lift:+.2f} | " f"[{lo:+.2f}, {hi:+.2f}] | {p:.4f} | " f"{r['filter_active_rate']:.1%} | {r['streams_per_week']:.1f} |" ) threshold = cfg["ship_threshold_pp"] lines += ["", "## Verdicts against the pre-registered rules", ""] for arm, (lift, p) in stats.items(): lines.append(f"- **{arm.upper()}**: {verdict(lift, p, threshold)} " f"(lift {lift:+.2f}pp against a +{threshold:.1f}pp bar)") b, c = results.get("b"), results.get("c") lines += ["", "## Secondary metrics", ""] if b: adoption = b["filter_active_rate"] bar = cfg["adoption_floor"] call = "clears" if adoption >= bar else "FAILS" lines.append( f"- Arm B adoption: {adoption:.1%}, {call} the pre-registered " f"{bar:.0%} floor. " + ("Demand is real, not just loud." if adoption >= bar else "The Reddit signal was louder than it was wide.") ) if c: averse_total = int(cfg["users_per_arm"] * cfg["segments"]["ai_averse"]["share"]) lines.append( f"- Arm C keeps the filter on for {c['filter_active_rate']:.1%} of users, " f"including {c['by_segment_active']['ai_averse']:,} of ~{averse_total:,} " f"AI-averse users." ) lines += [ "", "## The finding that matters", "", "Same feature, two defaults, different verdicts. The effect on users who " "actually run the filter is identical in both arms; what differs is how " "many users that is. **The default, not the feature, is the product " "decision.** An off-by-default toggle satisfies the vocal minority " "without moving the platform metric; on-by-default moves the metric and " "spends goodwill with the segment that wanted AI content.", "", "## What feeds back into the opportunity agent (stage 8)", "", "- Cluster C003 (AI-content trust) gets this experiment's verdict " "attached to its history.", "- If the shipped variant works, C003's complaint volume should shrink " "in subsequent weekly runs. The registry now makes that checkable: the " "trend line is the production follow-up to this experiment.", "- Next hypothesis raised by the data: targeted prompting (offer the " "filter to users who skip AI tracks) could capture most of arm C's " "lift without changing anyone's default.", "", "---", "*Generated by experiment.py. Change the assumptions in " "config/experiment.yaml and re-run; the verdicts recompute against the " "same pre-registered rules.*", ] out_path.write_text("\n".join(lines)) return stats def main(): parser = argparse.ArgumentParser() parser.add_argument("--config", default="config/experiment.yaml") parser.add_argument("--seed", type=int, default=42) args = parser.parse_args() cfg = yaml.safe_load(Path(args.config).read_text()) print(f"\nSimulating {len(cfg['arms'])} arms x {cfg['users_per_arm']:,} users " f"(SYNTHETIC, seed {args.seed})") results = simulate(cfg, args.seed) out_dir = Path("output") out_dir.mkdir(exist_ok=True) out_path = out_dir / "ab-test-readout.md" stats = write_readout(cfg, results, args.seed, out_path) for arm, (lift, p) in stats.items(): print(f" {arm.upper()}: lift {lift:+.2f}pp, p={p:.4f}") print(f"\nReadout: {out_path}\n") (out_dir / "ab-test-results.json").write_text(json.dumps(results, indent=2)) if __name__ == "__main__": main()