{"abstract":"Upper bounds are systematically too high.","category":"Experiment statistics","checks":8,"contract":"rng = random.Random(seed). Each of reps replicates resamples control then treatment with replacement (rng.choices, same sizes) and records mean(t) - mean(c). After sorting, the interval is diffs[floor((1 - level)/2 * reps)] to diffs[ceil((1 + level)/2 * reps) - 1]. Return both ends rounded to 6.","contract_signature":"control, treatment, reps, seed, level","evaluation_group":"w2-experiment-statistics-bootstrap-ci","failed_approach":"Flooring before subtracting one lands one position low for non-integral ranks.","family":"w2-experiment-statistics-bootstrap-ci-upper-index","id":"FA-74651","implementations":{"attempt":{"sha256":"9a4ddd2252fb5aa56dc157879a8c1b95f3904da97f5905acc56b2be1fca498d8","source":"\"\"\"Failure Map reference implementation. Python standard library only.\"\"\"\nimport json\nimport math\nimport random\nN = 1\nobservations = []\ndef solve(control, treatment, reps, seed, level):\n    rng = random.Random(seed)\n    diffs = []\n    for _ in range(reps):\n        c = rng.choices(control, k=len(control))\n        t = rng.choices(treatment, k=len(treatment))\n        diffs.append(sum(t) / len(t) - sum(c) / len(c))\n    diffs.sort()\n    lo = diffs[math.floor((1 - level) / 2 * reps)]\n    hi = diffs[math.floor((1 + level) / 2 * reps) - 1]\n    return [round(lo, 6), round(hi, 6)]\ndef check(label, actual, expected):\n    observations.append({\"check\": label, \"actual\": actual, \"expected\": expected, \"passed\": actual == expected})\nfixtures = [[('ninety percent interval', [[1, 2, 3, 4], [2, 3, 5, 8], 101, 7, 0.9], [0.0, 4.5]),\n  ('eighty percent interval', [[0, 0, 1, 5], [1, 1, 2], 50, 11, 0.8], [-2.333333, 1.083333]),\n  ('treatment larger than control', [[1, 2], [6, 9, 7], 41, 3, 0.9], [4.833333, 7.0]),\n  ('ninety-five percent interval', [[3, 1, 4, 1, 5], [9, 2, 6, 5], 200, 42, 0.95], [-0.2, 5.65]),\n  ('bootstrap sample 1', [[6, 3, 2, 7], [7, 9, 2, 9, 7], 200, 662, 0.9], [-0.35, 4.95]),\n  ('bootstrap sample 2', [[4, 4, 4, 8, 2], [10, 9, 6, 3, 2], 50, 548, 0.8], [-0.4, 3.0]),\n  ('bootstrap sample 3', [[6, 9, 2, 1], [3, 3, 11, 8, 2], 50, 269, 0.9], [-2.15, 4.35]),\n  ('bootstrap sample 4', [[4, 6, 4], [3, 11], 200, 10, 0.9], [-2.333333, 7.0])],\n [('ninety percent interval', [[1, 2, 3, 4], [2, 3, 5, 8], 101, 7, 0.9], [0.0, 4.5]),\n  ('eighty percent interval', [[0, 0, 1, 5], [1, 1, 2], 50, 11, 0.8], [-2.333333, 1.083333]),\n  ('treatment larger than control', [[1, 2], [6, 9, 7], 41, 3, 0.9], [4.833333, 7.0]),\n  ('ninety-five percent interval', [[3, 1, 4, 1, 5], [9, 2, 6, 5], 200, 42, 0.95], [-0.2, 5.65]),\n  ('bootstrap sample 6', [[9, 8], [11, 10, 6], 50, 172, 0.95], [-3.0, 2.5]),\n  ('bootstrap sample 7', [[5, 7, 2, 9], [1, 10], 41, 416, 0.9], [-6.0, 5.5]),\n  ('bootstrap sample 11', [[7, 5, 6, 7, 6], [6, 4, 6, 3], 41, 117, 0.9], [-2.9, -0.3]),\n  ('bootstrap sample 12', [[3, 5, 5, 0, 4], [5, 8], 41, 271, 0.95], [1.0, 5.6])],\n [('ninety percent interval', [[1, 2, 3, 4], [2, 3, 5, 8], 101, 7, 0.9], [0.0, 4.5]),\n  ('eighty percent interval', [[0, 0, 1, 5], [1, 1, 2], 50, 11, 0.8], [-2.333333, 1.083333]),\n  ('treatment larger than control', [[1, 2], [6, 9, 7], 41, 3, 0.9], [4.833333, 7.0]),\n  ('ninety-five percent interval', [[3, 1, 4, 1, 5], [9, 2, 6, 5], 200, 42, 0.95], [-0.2, 5.65]),\n  ('bootstrap sample 11', [[7, 5, 6, 7, 6], [6, 4, 6, 3], 41, 117, 0.9], [-2.9, -0.3]),\n  ('bootstrap sample 12', [[3, 5, 5, 0, 4], [5, 8], 41, 271, 0.95], [1.0, 5.6]),\n  ('bootstrap sample 24', [[1, 1, 3], [1, 8], 50, 923, 0.95], [-1.333333, 6.333333]),\n  ('bootstrap sample 32', [[2, 6, 6, 7], [3, 1, 2], 50, 938, 0.9], [-4.75, -1.666667])],\n [('ninety percent interval', [[1, 2, 3, 4], [2, 3, 5, 8], 101, 7, 0.9], [0.0, 4.5]),\n  ('eighty percent interval', [[0, 0, 1, 5], [1, 1, 2], 50, 11, 0.8], [-2.333333, 1.083333]),\n  ('treatment larger than control', [[1, 2], [6, 9, 7], 41, 3, 0.9], [4.833333, 7.0]),\n  ('ninety-five percent interval', [[3, 1, 4, 1, 5], [9, 2, 6, 5], 200, 42, 0.95], [-0.2, 5.65]),\n  ('bootstrap sample 16', [[3, 4], [5, 7, 7, 7], 101, 604, 0.95], [2.0, 4.0]),\n  ('bootstrap sample 17', [[0, 1], [10, 2, 5, 8, 11], 50, 644, 0.8], [4.5, 8.7]),\n  ('bootstrap sample 35', [[2, 0, 5, 8, 2], [5, 10, 2, 10], 50, 323, 0.8], [0.95, 5.75]),\n  ('bootstrap sample 52', [[1, 4, 4, 3, 1], [5, 9, 8], 101, 498, 0.95], [2.6, 6.733333])],\n [('ninety percent interval', [[1, 2, 3, 4], [2, 3, 5, 8], 101, 7, 0.9], [0.0, 4.5]),\n  ('eighty percent interval', [[0, 0, 1, 5], [1, 1, 2], 50, 11, 0.8], [-2.333333, 1.083333]),\n  ('treatment larger than control', [[1, 2], [6, 9, 7], 41, 3, 0.9], [4.833333, 7.0]),\n  ('ninety-five percent interval', [[3, 1, 4, 1, 5], [9, 2, 6, 5], 200, 42, 0.95], [-0.2, 5.65]),\n  ('bootstrap sample 9', [[2, 8, 6], [5, 2, 1, 11, 0], 50, 416, 0.9], [-5.266667, 3.6]),\n  ('bootstrap sample 21', [[2, 6, 8, 0, 8], [9, 10, 5, 7, 10], 41, 153, 0.8], [1.2, 5.2]),\n  ('bootstrap sample 22', [[8, 3], [0, 7, 11, 7, 1], 99, 38, 0.8], [-4.0, 3.4]),\n  ('bootstrap sample 50', [[4, 4, 9, 1, 8], [2, 1, 2, 0], 101, 193, 0.8], [-5.75, -2.1])]]\nfor label, args, expected in fixtures[N - 1]:\n    check(label, solve(*args), expected)\nprint(json.dumps({\"observations\": observations, \"passed\": all(x[\"passed\"] for x in observations)}, ensure_ascii=False))\nraise SystemExit(0 if all(x[\"passed\"] for x in observations) else 1)\n"},"broken":{"sha256":"cb79fae55022eb9d16d35ff302c2af4c6f810f0dd4777d884bd496e032308178","source":"\"\"\"Failure Map reference implementation. Python standard library only.\"\"\"\nimport json\nimport math\nimport random\nN = 1\nobservations = []\ndef solve(control, treatment, reps, seed, level):\n    rng = random.Random(seed)\n    diffs = []\n    for _ in range(reps):\n        c = rng.choices(control, k=len(control))\n        t = rng.choices(treatment, k=len(treatment))\n        diffs.append(sum(t) / len(t) - sum(c) / len(c))\n    diffs.sort()\n    lo = diffs[math.floor((1 - level) / 2 * reps)]\n    hi = diffs[math.ceil((1 + level) / 2 * reps)]\n    return [round(lo, 6), round(hi, 6)]\ndef check(label, actual, expected):\n    observations.append({\"check\": label, \"actual\": actual, \"expected\": expected, \"passed\": actual == expected})\nfixtures = [[('ninety percent interval', [[1, 2, 3, 4], [2, 3, 5, 8], 101, 7, 0.9], [0.0, 4.5]),\n  ('eighty percent interval', [[0, 0, 1, 5], [1, 1, 2], 50, 11, 0.8], [-2.333333, 1.083333]),\n  ('treatment larger than control', [[1, 2], [6, 9, 7], 41, 3, 0.9], [4.833333, 7.0]),\n  ('ninety-five percent interval', [[3, 1, 4, 1, 5], [9, 2, 6, 5], 200, 42, 0.95], [-0.2, 5.65]),\n  ('bootstrap sample 1', [[6, 3, 2, 7], [7, 9, 2, 9, 7], 200, 662, 0.9], [-0.35, 4.95]),\n  ('bootstrap sample 2', [[4, 4, 4, 8, 2], [10, 9, 6, 3, 2], 50, 548, 0.8], [-0.4, 3.0]),\n  ('bootstrap sample 3', [[6, 9, 2, 1], [3, 3, 11, 8, 2], 50, 269, 0.9], [-2.15, 4.35]),\n  ('bootstrap sample 4', [[4, 6, 4], [3, 11], 200, 10, 0.9], [-2.333333, 7.0])],\n [('ninety percent interval', [[1, 2, 3, 4], [2, 3, 5, 8], 101, 7, 0.9], [0.0, 4.5]),\n  ('eighty percent interval', [[0, 0, 1, 5], [1, 1, 2], 50, 11, 0.8], [-2.333333, 1.083333]),\n  ('treatment larger than control', [[1, 2], [6, 9, 7], 41, 3, 0.9], [4.833333, 7.0]),\n  ('ninety-five percent interval', [[3, 1, 4, 1, 5], [9, 2, 6, 5], 200, 42, 0.95], [-0.2, 5.65]),\n  ('bootstrap sample 6', [[9, 8], [11, 10, 6], 50, 172, 0.95], [-3.0, 2.5]),\n  ('bootstrap sample 7', [[5, 7, 2, 9], [1, 10], 41, 416, 0.9], [-6.0, 5.5]),\n  ('bootstrap sample 11', [[7, 5, 6, 7, 6], [6, 4, 6, 3], 41, 117, 0.9], [-2.9, -0.3]),\n  ('bootstrap sample 12', [[3, 5, 5, 0, 4], [5, 8], 41, 271, 0.95], [1.0, 5.6])],\n [('ninety percent interval', [[1, 2, 3, 4], [2, 3, 5, 8], 101, 7, 0.9], [0.0, 4.5]),\n  ('eighty percent interval', [[0, 0, 1, 5], [1, 1, 2], 50, 11, 0.8], [-2.333333, 1.083333]),\n  ('treatment larger than control', [[1, 2], [6, 9, 7], 41, 3, 0.9], [4.833333, 7.0]),\n  ('ninety-five percent interval', [[3, 1, 4, 1, 5], [9, 2, 6, 5], 200, 42, 0.95], [-0.2, 5.65]),\n  ('bootstrap sample 11', [[7, 5, 6, 7, 6], [6, 4, 6, 3], 41, 117, 0.9], [-2.9, -0.3]),\n  ('bootstrap sample 12', [[3, 5, 5, 0, 4], [5, 8], 41, 271, 0.95], [1.0, 5.6]),\n  ('bootstrap sample 24', [[1, 1, 3], [1, 8], 50, 923, 0.95], [-1.333333, 6.333333]),\n  ('bootstrap sample 32', [[2, 6, 6, 7], [3, 1, 2], 50, 938, 0.9], [-4.75, -1.666667])],\n [('ninety percent interval', [[1, 2, 3, 4], [2, 3, 5, 8], 101, 7, 0.9], [0.0, 4.5]),\n  ('eighty percent interval', [[0, 0, 1, 5], [1, 1, 2], 50, 11, 0.8], [-2.333333, 1.083333]),\n  ('treatment larger than control', [[1, 2], [6, 9, 7], 41, 3, 0.9], [4.833333, 7.0]),\n  ('ninety-five percent interval', [[3, 1, 4, 1, 5], [9, 2, 6, 5], 200, 42, 0.95], [-0.2, 5.65]),\n  ('bootstrap sample 16', [[3, 4], [5, 7, 7, 7], 101, 604, 0.95], [2.0, 4.0]),\n  ('bootstrap sample 17', [[0, 1], [10, 2, 5, 8, 11], 50, 644, 0.8], [4.5, 8.7]),\n  ('bootstrap sample 35', [[2, 0, 5, 8, 2], [5, 10, 2, 10], 50, 323, 0.8], [0.95, 5.75]),\n  ('bootstrap sample 52', [[1, 4, 4, 3, 1], [5, 9, 8], 101, 498, 0.95], [2.6, 6.733333])],\n [('ninety percent interval', [[1, 2, 3, 4], [2, 3, 5, 8], 101, 7, 0.9], [0.0, 4.5]),\n  ('eighty percent interval', [[0, 0, 1, 5], [1, 1, 2], 50, 11, 0.8], [-2.333333, 1.083333]),\n  ('treatment larger than control', [[1, 2], [6, 9, 7], 41, 3, 0.9], [4.833333, 7.0]),\n  ('ninety-five percent interval', [[3, 1, 4, 1, 5], [9, 2, 6, 5], 200, 42, 0.95], [-0.2, 5.65]),\n  ('bootstrap sample 9', [[2, 8, 6], [5, 2, 1, 11, 0], 50, 416, 0.9], [-5.266667, 3.6]),\n  ('bootstrap sample 21', [[2, 6, 8, 0, 8], [9, 10, 5, 7, 10], 41, 153, 0.8], [1.2, 5.2]),\n  ('bootstrap sample 22', [[8, 3], [0, 7, 11, 7, 1], 99, 38, 0.8], [-4.0, 3.4]),\n  ('bootstrap sample 50', [[4, 4, 9, 1, 8], [2, 1, 2, 0], 101, 193, 0.8], [-5.75, -2.1])]]\nfor label, args, expected in fixtures[N - 1]:\n    check(label, solve(*args), expected)\nprint(json.dumps({\"observations\": observations, \"passed\": all(x[\"passed\"] for x in observations)}, ensure_ascii=False))\nraise SystemExit(0 if all(x[\"passed\"] for x in observations) else 1)\n"}},"limitations":"A deterministic toy experiment-analysis model with a stipulated contract; results are rounded and are not a substitute for a validated statistics package. This reproducer isolates one failure mechanism. Results cover the supplied fixtures. Variants within a family share a test contract and should remain grouped when constructing evaluation splits. Related mechanisms with a shared evaluation_group must also remain together; these controlled models are not independent production incidents.","method":"Deterministic executable model with adversarial boundary fixtures.","provenance":{"created_by":"Failure Map","dependencies":"Python standard library","family":"w2-experiment-statistics-bootstrap-ci-upper-index","generated_at":"2026-09-29T14:48:58.782686+00:00","license":"CC0-1.0","python":"3.12.14","seed":1,"split":"open-access"},"relevance":"Bootstrap intervals are the fallback for skewed metrics; reproducibility and indexing must be exact.","root_cause":"The upper index omits the -1 conversion from a count to a position.","sha256":"f52ff2a15657a9ca344e169e748d5038342bcacc9eb13f3c2a3a895afd0f63bb","title":"Bootstrap difference interval: The upper percentile reads one position too far · case 01","variant":1,"variant_policy":"Five numbered records share a model and may reuse boundary fixtures.","verified":true,"visibility":"public","verification":{"attempt":{"elapsed_ms":41.454,"exit_code":1,"observations":[{"actual":[0.0,4.25],"check":"ninety percent interval","expected":[0.0,4.5],"passed":false},{"actual":[-2.333333,1.083333],"check":"eighty percent interval","expected":[-2.333333,1.083333],"passed":true},{"actual":[4.833333,7.0],"check":"treatment larger than control","expected":[4.833333,7.0],"passed":true},{"actual":[-0.2,5.65],"check":"ninety-five percent interval","expected":[-0.2,5.65],"passed":true},{"actual":[-0.35,4.95],"check":"bootstrap sample 1","expected":[-0.35,4.95],"passed":true},{"actual":[-0.4,3.0],"check":"bootstrap sample 2","expected":[-0.4,3.0],"passed":true},{"actual":[-2.15,3.95],"check":"bootstrap sample 3","expected":[-2.15,4.35],"passed":false},{"actual":[-2.333333,7.0],"check":"bootstrap sample 4","expected":[-2.333333,7.0],"passed":true}],"passed":false,"stderr":"","stdout":"{\"observations\": [{\"check\": \"ninety percent interval\", \"actual\": [0.0, 4.25], \"expected\": [0.0, 4.5], \"passed\": false}, {\"check\": \"eighty percent interval\", \"actual\": [-2.333333, 1.083333], \"expected\": [-2.333333, 1.083333], \"passed\": true}, {\"check\": \"treatment larger than control\", \"actual\": [4.833333, 7.0], \"expected\": [4.833333, 7.0], \"passed\": true}, {\"check\": \"ninety-five percent interval\", \"actual\": [-0.2, 5.65], \"expected\": [-0.2, 5.65], \"passed\": true}, {\"check\": \"bootstrap sample 1\", \"actual\": [-0.35, 4.95], \"expected\": [-0.35, 4.95], \"passed\": true}, {\"check\": \"bootstrap sample 2\", \"actual\": [-0.4, 3.0], \"expected\": [-0.4, 3.0], \"passed\": true}, {\"check\": \"bootstrap sample 3\", \"actual\": [-2.15, 3.95], \"expected\": [-2.15, 4.35], \"passed\": false}, {\"check\": \"bootstrap sample 4\", \"actual\": [-2.333333, 7.0], \"expected\": [-2.333333, 7.0], \"passed\": true}], \"passed\": false}\n"},"broken":{"elapsed_ms":43.418,"exit_code":1,"observations":[{"actual":[0.0,4.5],"check":"ninety percent interval","expected":[0.0,4.5],"passed":true},{"actual":[-2.333333,1.083333],"check":"eighty percent interval","expected":[-2.333333,1.083333],"passed":true},{"actual":[4.833333,7.333333],"check":"treatment larger than control","expected":[4.833333,7.0],"passed":false},{"actual":[-0.2,5.9],"check":"ninety-five percent interval","expected":[-0.2,5.65],"passed":false},{"actual":[-0.35,4.95],"check":"bootstrap sample 1","expected":[-0.35,4.95],"passed":true},{"actual":[-0.4,3.0],"check":"bootstrap sample 2","expected":[-0.4,3.0],"passed":true},{"actual":[-2.15,4.7],"check":"bootstrap sample 3","expected":[-2.15,4.35],"passed":false},{"actual":[-2.333333,7.0],"check":"bootstrap sample 4","expected":[-2.333333,7.0],"passed":true}],"passed":false,"stderr":"","stdout":"{\"observations\": [{\"check\": \"ninety percent interval\", \"actual\": [0.0, 4.5], \"expected\": [0.0, 4.5], \"passed\": true}, {\"check\": \"eighty percent interval\", \"actual\": [-2.333333, 1.083333], \"expected\": [-2.333333, 1.083333], \"passed\": true}, {\"check\": \"treatment larger than control\", \"actual\": [4.833333, 7.333333], \"expected\": [4.833333, 7.0], \"passed\": false}, {\"check\": \"ninety-five percent interval\", \"actual\": [-0.2, 5.9], \"expected\": [-0.2, 5.65], \"passed\": false}, {\"check\": \"bootstrap sample 1\", \"actual\": [-0.35, 4.95], \"expected\": [-0.35, 4.95], \"passed\": true}, {\"check\": \"bootstrap sample 2\", \"actual\": [-0.4, 3.0], \"expected\": [-0.4, 3.0], \"passed\": true}, {\"check\": \"bootstrap sample 3\", \"actual\": [-2.15, 4.7], \"expected\": [-2.15, 4.35], \"passed\": false}, {\"check\": \"bootstrap sample 4\", \"actual\": [-2.333333, 7.0], \"expected\": [-2.333333, 7.0], \"passed\": true}], \"passed\": false}\n"}},"member_only":{"stages":["fixed"],"fields":["implementations.fixed","verification.fixed","harness","repair"],"note":"The verified repair, its recorded checks, the repair description, and the scoring harness are available to members."}}