Repository navigation
Expand file tree
/
Copy pathpeeking_simulation.py
More file actions
89 lines (69 loc) · 2.9 KB
/
Copy pathpeeking_simulation.py
File metadata and controls
89 lines (69 loc) · 2.9 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
"""
The "peeking" problem — why stopping early inflates false positives
==================================================================
A very common A/B mistake: watching the test live and stopping the moment it
looks significant. Each look is another chance to cross p < 0.05 by luck, so the
real false-positive rate balloons far above the intended 5%.
We demonstrate it with A/A tests (no true effect): compare a single look at the
end vs. peeking at several interim points and stopping early on significance.
Run: python peeking_simulation.py
Prints both false-positive rates and saves a comparison chart.
Author: Muhammad Nasiruddin
"""
from __future__ import annotations
from pathlib import Path
import matplotlib
matplotlib.use("Agg")
import matplotlib.pyplot as plt
import numpy as np
from statsmodels.stats.proportion import proportions_ztest
IMG = Path(__file__).parent / "images"
WARN, ACCENT = "#e76f51", "#2a9d8f"
RNG = np.random.default_rng(202)
N_SIMS = 3_000
N_FINAL = 6_000 # visitors per arm at the end
LOOKS = list(range(1000, N_FINAL + 1, 1000)) # peek every 1,000
BASE_RATE = 0.12
ALPHA = 0.05
def _pval(ca: int, na: int, cb: int, nb: int) -> float:
if ca + cb == 0:
return 1.0
return proportions_ztest([ca, cb], [na, nb])[1]
def main() -> None:
single_hits = 0
peek_hits = 0
for _ in range(N_SIMS):
# Per-visitor conversions for two identical arms (A/A).
a = RNG.binomial(1, BASE_RATE, N_FINAL)
b = RNG.binomial(1, BASE_RATE, N_FINAL)
ca, cb = np.cumsum(a), np.cumsum(b)
# Honest single look at the end.
if _pval(ca[-1], N_FINAL, cb[-1], N_FINAL) < ALPHA:
single_hits += 1
# Peeking: stop as soon as any interim look is significant.
for m in LOOKS:
if _pval(ca[m - 1], m, cb[m - 1], m) < ALPHA:
peek_hits += 1
break
fpr_single = single_hits / N_SIMS
fpr_peek = peek_hits / N_SIMS
print(f"A/A tests: {N_SIMS:,} sims, {len(LOOKS)} interim looks.")
print(f" single look at end: false-positive rate {fpr_single:.1%} "
f"(target {ALPHA:.0%})")
print(f" peeking + early stopping: false-positive rate {fpr_peek:.1%} "
f"({fpr_peek/ALPHA:.1f}x the intended rate!)")
print(" Lesson: fix the sample size in advance, or use a sequential test.\n")
fig, ax = plt.subplots(figsize=(6, 4))
ax.bar(["Single look\n(correct)", "Peeking\n(early stop)"],
[fpr_single * 100, fpr_peek * 100], color=[ACCENT, WARN])
ax.axhline(ALPHA * 100, color="black", linestyle="--", linewidth=1,
label="intended 5%")
ax.set_ylabel("False-positive rate (%)")
ax.set_title("Peeking inflates false positives")
ax.legend()
fig.tight_layout()
fig.savefig(IMG / "peeking.png", dpi=110)
plt.close(fig)
print("Saved chart to images/peeking.png")
if __name__ == "__main__":
main()