{ "name": "ab-test-results", "type": "registry:ui", "title": "A/B Test Results", "description": "Variants with lift, a confidence interval and a two-proportion z-test — plus the peeking warning, because stopping at the first significant reading is the failure that actually happens.", "category": "Analytics", "dependencies": [ "lucide-react@^1.39.0" ], "registryDependencies": [ "lib-motion", "lib-styles", "lib-utils", "table" ], "files": [ { "path": "components/ui/ab-test-results.tsx", "type": "registry:ui", "content": "'use client'\n\nimport { useId, useMemo, type ComponentProps, type ReactNode } from 'react'\nimport { AlertTriangle } from 'lucide-react'\nimport {\n TableBody,\n TableCell,\n TableHead,\n TableHeader,\n TableRow,\n} from '@/components/ui/table'\nimport { enterFade } from '@/lib/motion'\nimport { radius, surface } from '@/lib/styles'\nimport { cn } from '@/lib/utils'\n\n/**\n * Experiment variants with lift, a confidence interval, and an honest verdict.\n *\n * **It shows the interval, not just the number.** \"+12% lift\" is not a result;\n * \"+12%, 95% CI [−3%, +27%]\" is, and it says the opposite — that the experiment\n * has not concluded. A dashboard that prints the point estimate alone\n * manufactures certainty, and people ship on it.\n *\n * **The maths.** A two-proportion z-test with a pooled standard error for the\n * p-value, and an unpooled standard error for the interval — the standard pair,\n * and the reason the interval can straddle zero while p sits just under 0.05.\n * The interval is then divided by the control rate so it is on the same scale\n * as the lift beside it; an absolute-difference interval printed next to a\n * relative lift is read as bounding that lift, and it does not.\n * The normal approximation needs roughly ten conversions in each arm, so below\n * that the verdict is withheld rather than computed on data too thin to carry\n * it.\n *\n * **It warns about peeking, because that is the real-world failure.** Checking\n * an experiment repeatedly and stopping at the first significant reading\n * inflates the false-positive rate far above the nominal 5% — a fixed-horizon\n * test read continuously is wrong roughly a third of the time. If a sample-size\n * target is given and has not been reached, this says so plainly.\n *\n * It reports a *statistical* result about one metric. Whether that metric is\n * the one that matters, whether the assignment was actually random, and whether\n * the effect is worth the change are not things any component can tell you.\n */\nexport type Variant = {\n id: string\n name: ReactNode\n /** Users in this arm. */\n visitors: number\n conversions: number\n /** Exactly one variant should be the baseline. */\n control?: boolean\n}\n\ntype AbTestResultsProps = Omit, 'title'> & {\n variants: Variant[]\n title?: ReactNode\n /** Two-sided significance threshold. */\n alpha?: number\n /** Per-arm sample size the test was planned for. Enables the peeking warning. */\n targetSample?: number\n metricLabel?: string\n emptyLabel?: string\n label?: string\n}\n\n/** Normal CDF via Abramowitz & Stegun 7.1.26 — accurate to ~1e-7. */\nfunction normalCdf(z: number): number {\n const sign = z < 0 ? -1 : 1\n const x = Math.abs(z) / Math.SQRT2\n const t = 1 / (1 + 0.3275911 * x)\n const y =\n 1 -\n ((((1.061405429 * t - 1.453152027) * t + 1.421413741) * t - 0.284496736) * t + 0.254829592) *\n t *\n Math.exp(-x * x)\n return 0.5 * (1 + sign * y)\n}\n\n/** 95% ≈ 1.96. Inverse normal by bisection: exact enough, and tiny. */\nfunction zFor(confidence: number): number {\n let lo = 0\n let hi = 6\n const target = 1 - (1 - confidence) / 2\n for (let i = 0; i < 60; i++) {\n const mid = (lo + hi) / 2\n if (normalCdf(mid) < target) lo = mid\n else hi = mid\n }\n return (lo + hi) / 2\n}\n\nfunction AbTestResults({\n variants,\n title,\n alpha = 0.05,\n targetSample,\n metricLabel = 'Conversion',\n emptyLabel = 'No variants.',\n label = 'Experiment results',\n className,\n ...props\n}: AbTestResultsProps) {\n const titleId = useId()\n\n const rows = useMemo(() => {\n const control = variants.find((variant) => variant.control) ?? variants[0]\n if (!control) return []\n\n const z = zFor(1 - alpha)\n const controlRate = control.visitors > 0 ? control.conversions / control.visitors : 0\n\n return variants.map((variant) => {\n const rate = variant.visitors > 0 ? variant.conversions / variant.visitors : 0\n if (variant.id === control.id) {\n return { variant, rate, isControl: true, lift: null, ci: null, p: null, thin: false }\n }\n\n // The normal approximation needs enough successes and failures in both\n // arms; under that, no verdict is honest.\n const thin =\n Math.min(\n variant.conversions,\n variant.visitors - variant.conversions,\n control.conversions,\n control.visitors - control.conversions,\n ) < 10\n\n const pooled =\n (variant.conversions + control.conversions) / (variant.visitors + control.visitors)\n const pooledSe = Math.sqrt(\n pooled * (1 - pooled) * (1 / variant.visitors + 1 / control.visitors),\n )\n const statistic = pooledSe > 0 ? (rate - controlRate) / pooledSe : 0\n const p = 2 * (1 - normalCdf(Math.abs(statistic)))\n\n // Unpooled for the interval — pooling assumes the null it is testing.\n const se = Math.sqrt(\n (rate * (1 - rate)) / variant.visitors + (controlRate * (1 - controlRate)) / control.visitors,\n )\n const diff = rate - controlRate\n\n /**\n * The interval is put on the same scale as the lift beside it.\n *\n * The z-test gives an interval for the *absolute* difference in rates —\n * 0.75 percentage points. Printing that next to a *relative* lift of\n * +12.5% invites the reader to take it as bounding the lift, which it\n * does not: the two numbers are in different units and disagree by a\n * factor of the control rate. Dividing by the control rate is the\n * first-order (delta method) interval for relative lift, which is what\n * every experiment dashboard means by \"95% CI\" on a lift column.\n */\n const ci: [number, number] =\n controlRate > 0\n ? [(diff - z * se) / controlRate, (diff + z * se) / controlRate]\n : [diff - z * se, diff + z * se]\n\n return {\n variant,\n rate,\n isControl: false,\n lift: controlRate > 0 ? diff / controlRate : null,\n ci,\n p,\n thin,\n }\n })\n }, [variants, alpha])\n\n if (rows.length === 0) {\n return (\n
\n

{emptyLabel}

\n
\n )\n }\n\n const underpowered =\n targetSample !== undefined && variants.some((variant) => variant.visitors < targetSample)\n\n const percent = (value: number) => `${(value * 100).toFixed(2)}%`\n const signed = (value: number) => `${value >= 0 ? '+' : ''}${(value * 100).toFixed(1)}%`\n\n return (\n \n {title && (\n
\n

\n {title}\n

\n
\n )}\n {!title && (\n

\n {label}\n

\n )}\n\n
\n {/* The kit's table parts: the six column headings were six copies of\n the same recipe, and the rows now pick up the shared rule and hover\n for free. */}\n \n \n \n Variant\n Visitors\n {metricLabel}\n Lift\n \n {Math.round((1 - alpha) * 100)}% CI on lift\n \n p\n \n \n \n {rows.map((row) => {\n const significant = row.p !== null && !row.thin && row.p < alpha\n return (\n \n \n \n {row.variant.name}\n {row.isControl && (\n control\n )}\n \n \n \n {row.variant.visitors.toLocaleString()}\n \n \n {percent(row.rate)}\n \n ({row.variant.conversions})\n \n \n 0 && 'text-[var(--green-soft-foreground)]',\n significant && (row.lift ?? 0) < 0 && 'text-[var(--destructive)]',\n )}\n >\n {row.lift === null ? '—' : signed(row.lift)}\n \n \n {row.ci ? `[${signed(row.ci[0])}, ${signed(row.ci[1])}]` : '—'}\n \n \n {row.p === null ? (\n '—'\n ) : row.thin ? (\n too few\n ) : (\n \n {row.p < 0.001 ? '<0.001' : row.p.toFixed(3)}\n \n )}\n \n \n )\n })}\n \n
\n
\n\n {/* The warning that matters more than the p-value. */}\n {underpowered && (\n \n \n \n Below the planned {targetSample?.toLocaleString()} per arm. Stopping at the first\n significant reading inflates the false-positive rate well past {Math.round(alpha * 100)}%\n — treat anything here as provisional.\n \n

\n )}\n \n )\n}\n\nexport { AbTestResults }\nexport type { AbTestResultsProps }\n" } ] }