{"abstract":"Balanced data produces strongly negative z-scores.","category":"Experiment statistics","checks":8,"contract":"Pool a (control) and b (treatment), assign mid-ranks to ties (1-based). U = R_b - n_b(n_b + 1)/2. Tie-corrected variance = n_a n_b / 12 * ((N + 1) - sum(t^3 - t) / (N(N - 1))); z = (U - n_a n_b / 2) / sqrt(variance) without continuity correction; zero variance gives z = 0. Empty arm -> None. Return [U, round(z, 6)].","contract_signature":"a, b","evaluation_group":"w2-experiment-statistics-mann-whitney","failed_approach":"Centring at (n_a + n_b) / 2 confuses the U scale with the rank scale.","family":"w2-experiment-statistics-mann-whitney-null-center","id":"FA-74686","implementations":{"attempt":{"sha256":"7bb44188fe4b2be1b6ab7c8cbb99873debd7a598fd53c13631461f7b38b7314c","source":"\"\"\"Failure Map reference implementation. Python standard library only.\"\"\"\nimport json\nimport math\nN = 1\nobservations = []\ndef solve(a, b):\n    na, nb = len(a), len(b)\n    if na == 0 or nb == 0:\n        return None\n    pooled = sorted([(v, 0) for v in a] + [(v, 1) for v in b])\n    N = na + nb\n    ranks = [0.0] * N\n    ties = 0\n    i = 0\n    while i < N:\n        j = i\n        while j + 1 < N and pooled[j + 1][0] == pooled[i][0]:\n            j += 1\n        mid = (i + j) / 2 + 1\n        for k in range(i, j + 1):\n            ranks[k] = mid\n        t = j - i + 1\n        ties += t ** 3 - t\n        i = j + 1\n    rb = sum(r for r, (v, g) in zip(ranks, pooled) if g == 1)\n    u = rb - nb * (nb + 1) / 2\n    var = na * nb / 12 * ((N + 1) - ties / (N * (N - 1))) if N > 1 else 0.0\n    if var <= 0:\n        return [u, 0.0]\n    return [u, round((u - (na + nb) / 2) / math.sqrt(var), 6)]\ndef check(label, actual, expected):\n    observations.append({\"check\": label, \"actual\": actual, \"expected\": expected, \"passed\": actual == expected})\nfixtures = [[('ties get average ranks', [[1, 2, 2], [2, 3, 4]], [8.0, 1.623086]),\n  ('two-way tie across arms', [[1, 2], [2, 3]], [3.5, 1.224745]),\n  ('three-way tie with treatment', [[5, 5], [5, 6]], [3.0, 1.0]),\n  ('no ties', [[1, 3, 5], [2, 4, 6, 8]], [9.0, 1.06066]),\n  ('complete separation', [[1, 2], [3, 4, 5]], [6.0, 1.732051]),\n  ('unequal arm sizes', [[0, 1, 1, 4], [1, 2]], [5.0, 0.491869]),\n  ('rank sample 1', [[1, 3, 2], [7]], [3.0, 1.341641]),\n  ('rank sample 2', [[3, 1, 2, 2, 4], [7, 3, 1, 1, 5, 3]], [18.0, 0.559282])],\n [('two-way tie across arms', [[1, 2], [2, 3]], [3.5, 1.224745]),\n  ('three-way tie with treatment', [[5, 5], [5, 6]], [3.0, 1.0]),\n  ('no ties', [[1, 3, 5], [2, 4, 6, 8]], [9.0, 1.06066]),\n  ('complete separation', [[1, 2], [3, 4, 5]], [6.0, 1.732051]),\n  ('all values tied', [[2, 2], [2, 2, 2]], [3.0, 0.0]),\n  ('unequal arm sizes', [[0, 1, 1, 4], [1, 2]], [5.0, 0.491869]),\n  ('rank sample 2', [[3, 1, 2, 2, 4], [7, 3, 1, 1, 5, 3]], [18.0, 0.559282]),\n  ('rank sample 4', [[4], [3, 4, 2, 7, 2, 3]], [1.5, -0.770934])],\n [('three-way tie with treatment', [[5, 5], [5, 6]], [3.0, 1.0]),\n  ('no ties', [[1, 3, 5], [2, 4, 6, 8]], [9.0, 1.06066]),\n  ('complete separation', [[1, 2], [3, 4, 5]], [6.0, 1.732051]),\n  ('all values tied', [[2, 2], [2, 2, 2]], [3.0, 0.0]),\n  ('unequal arm sizes', [[0, 1, 1, 4], [1, 2]], [5.0, 0.491869]),\n  ('rank sample 9', [[5, 3, 4], [1, 1]], [0.0, -1.777047]),\n  ('rank sample 11', [[1, 1, 2, 3, 4], [2]], [2.5, 0.0]),\n  ('rank sample 12', [[2, 1, 4], [6, 0, 3, 7]], [8.0, 0.707107])],\n [('ties get average ranks', [[1, 2, 2], [2, 3, 4]], [8.0, 1.623086]),\n  ('no ties', [[1, 3, 5], [2, 4, 6, 8]], [9.0, 1.06066]),\n  ('complete separation', [[1, 2], [3, 4, 5]], [6.0, 1.732051]),\n  ('all values tied', [[2, 2], [2, 2, 2]], [3.0, 0.0]),\n  ('unequal arm sizes', [[0, 1, 1, 4], [1, 2]], [5.0, 0.491869]),\n  ('rank sample 16', [[4, 0, 3, 3, 1], [4, 1]], [6.0, 0.398109]),\n  ('rank sample 17', [[3, 5, 0], [0, 6, 2, 4, 4, 6]], [11.5, 0.65372]),\n  ('rank sample 19', [[1, 3, 3, 0], [6, 5]], [8.0, 1.878673])],\n [('ties get average ranks', [[1, 2, 2], [2, 3, 4]], [8.0, 1.623086]),\n  ('two-way tie across arms', [[1, 2], [2, 3]], [3.5, 1.224745]),\n  ('complete separation', [[1, 2], [3, 4, 5]], [6.0, 1.732051]),\n  ('all values tied', [[2, 2], [2, 2, 2]], [3.0, 0.0]),\n  ('unequal arm sizes', [[0, 1, 1, 4], [1, 2]], [5.0, 0.491869]),\n  ('rank sample 21', [[4, 4, 3, 1, 2], [4, 7, 7, 7, 6, 5]], [29.0, 2.603819]),\n  ('rank sample 23', [[3, 4], [0, 4, 6]], [3.5, 0.296174]),\n  ('rank sample 27', [[2, 4, 4, 4], [1, 6]], [4.0, 0.0])]]\nfor label, args, expected in fixtures[N - 1]:\n    check(label, solve(*args), expected)\nprint(json.dumps({\"observations\": observations, \"passed\": all(x[\"passed\"] for x in observations)}, ensure_ascii=False))\nraise SystemExit(0 if all(x[\"passed\"] for x in observations) else 1)\n"},"broken":{"sha256":"7e923b176b94d8a361748035cdcfb258a41d9f3beddbf7c870bf60f06dbcf755","source":"\"\"\"Failure Map reference implementation. Python standard library only.\"\"\"\nimport json\nimport math\nN = 1\nobservations = []\ndef solve(a, b):\n    na, nb = len(a), len(b)\n    if na == 0 or nb == 0:\n        return None\n    pooled = sorted([(v, 0) for v in a] + [(v, 1) for v in b])\n    N = na + nb\n    ranks = [0.0] * N\n    ties = 0\n    i = 0\n    while i < N:\n        j = i\n        while j + 1 < N and pooled[j + 1][0] == pooled[i][0]:\n            j += 1\n        mid = (i + j) / 2 + 1\n        for k in range(i, j + 1):\n            ranks[k] = mid\n        t = j - i + 1\n        ties += t ** 3 - t\n        i = j + 1\n    rb = sum(r for r, (v, g) in zip(ranks, pooled) if g == 1)\n    u = rb - nb * (nb + 1) / 2\n    var = na * nb / 12 * ((N + 1) - ties / (N * (N - 1))) if N > 1 else 0.0\n    if var <= 0:\n        return [u, 0.0]\n    return [u, round((u - na * nb) / math.sqrt(var), 6)]\ndef check(label, actual, expected):\n    observations.append({\"check\": label, \"actual\": actual, \"expected\": expected, \"passed\": actual == expected})\nfixtures = [[('ties get average ranks', [[1, 2, 2], [2, 3, 4]], [8.0, 1.623086]),\n  ('two-way tie across arms', [[1, 2], [2, 3]], [3.5, 1.224745]),\n  ('three-way tie with treatment', [[5, 5], [5, 6]], [3.0, 1.0]),\n  ('no ties', [[1, 3, 5], [2, 4, 6, 8]], [9.0, 1.06066]),\n  ('complete separation', [[1, 2], [3, 4, 5]], [6.0, 1.732051]),\n  ('unequal arm sizes', [[0, 1, 1, 4], [1, 2]], [5.0, 0.491869]),\n  ('rank sample 1', [[1, 3, 2], [7]], [3.0, 1.341641]),\n  ('rank sample 2', [[3, 1, 2, 2, 4], [7, 3, 1, 1, 5, 3]], [18.0, 0.559282])],\n [('two-way tie across arms', [[1, 2], [2, 3]], [3.5, 1.224745]),\n  ('three-way tie with treatment', [[5, 5], [5, 6]], [3.0, 1.0]),\n  ('no ties', [[1, 3, 5], [2, 4, 6, 8]], [9.0, 1.06066]),\n  ('complete separation', [[1, 2], [3, 4, 5]], [6.0, 1.732051]),\n  ('all values tied', [[2, 2], [2, 2, 2]], [3.0, 0.0]),\n  ('unequal arm sizes', [[0, 1, 1, 4], [1, 2]], [5.0, 0.491869]),\n  ('rank sample 2', [[3, 1, 2, 2, 4], [7, 3, 1, 1, 5, 3]], [18.0, 0.559282]),\n  ('rank sample 4', [[4], [3, 4, 2, 7, 2, 3]], [1.5, -0.770934])],\n [('three-way tie with treatment', [[5, 5], [5, 6]], [3.0, 1.0]),\n  ('no ties', [[1, 3, 5], [2, 4, 6, 8]], [9.0, 1.06066]),\n  ('complete separation', [[1, 2], [3, 4, 5]], [6.0, 1.732051]),\n  ('all values tied', [[2, 2], [2, 2, 2]], [3.0, 0.0]),\n  ('unequal arm sizes', [[0, 1, 1, 4], [1, 2]], [5.0, 0.491869]),\n  ('rank sample 9', [[5, 3, 4], [1, 1]], [0.0, -1.777047]),\n  ('rank sample 11', [[1, 1, 2, 3, 4], [2]], [2.5, 0.0]),\n  ('rank sample 12', [[2, 1, 4], [6, 0, 3, 7]], [8.0, 0.707107])],\n [('ties get average ranks', [[1, 2, 2], [2, 3, 4]], [8.0, 1.623086]),\n  ('no ties', [[1, 3, 5], [2, 4, 6, 8]], [9.0, 1.06066]),\n  ('complete separation', [[1, 2], [3, 4, 5]], [6.0, 1.732051]),\n  ('all values tied', [[2, 2], [2, 2, 2]], [3.0, 0.0]),\n  ('unequal arm sizes', [[0, 1, 1, 4], [1, 2]], [5.0, 0.491869]),\n  ('rank sample 16', [[4, 0, 3, 3, 1], [4, 1]], [6.0, 0.398109]),\n  ('rank sample 17', [[3, 5, 0], [0, 6, 2, 4, 4, 6]], [11.5, 0.65372]),\n  ('rank sample 19', [[1, 3, 3, 0], [6, 5]], [8.0, 1.878673])],\n [('ties get average ranks', [[1, 2, 2], [2, 3, 4]], [8.0, 1.623086]),\n  ('two-way tie across arms', [[1, 2], [2, 3]], [3.5, 1.224745]),\n  ('complete separation', [[1, 2], [3, 4, 5]], [6.0, 1.732051]),\n  ('all values tied', [[2, 2], [2, 2, 2]], [3.0, 0.0]),\n  ('unequal arm sizes', [[0, 1, 1, 4], [1, 2]], [5.0, 0.491869]),\n  ('rank sample 21', [[4, 4, 3, 1, 2], [4, 7, 7, 7, 6, 5]], [29.0, 2.603819]),\n  ('rank sample 23', [[3, 4], [0, 4, 6]], [3.5, 0.296174]),\n  ('rank sample 27', [[2, 4, 4, 4], [1, 6]], [4.0, 0.0])]]\nfor label, args, expected in fixtures[N - 1]:\n    check(label, solve(*args), expected)\nprint(json.dumps({\"observations\": observations, \"passed\": all(x[\"passed\"] for x in observations)}, ensure_ascii=False))\nraise SystemExit(0 if all(x[\"passed\"] for x in observations) else 1)\n"}},"limitations":"A deterministic toy experiment-analysis model with a stipulated contract; results are rounded and are not a substitute for a validated statistics package. This reproducer isolates one failure mechanism. Results cover the supplied fixtures. Variants within a family share a test contract and should remain grouped when constructing evaluation splits. Related mechanisms with a shared evaluation_group must also remain together; these controlled models are not independent production incidents.","method":"Deterministic executable model with adversarial boundary fixtures.","provenance":{"created_by":"Failure Map","dependencies":"Python standard library","family":"w2-experiment-statistics-mann-whitney-null-center","generated_at":"2026-09-29T14:48:59.069992+00:00","license":"CC0-1.0","python":"3.12.14","seed":1,"split":"open-access"},"relevance":"Rank tests are used for heavy-tailed metrics such as latency; ties are common in bucketed data.","root_cause":"The null expectation of U is taken as n_a n_b instead of n_a n_b / 2.","sha256":"f537bf188a85f38dbf4c7d511e415ebb7224812d82d13bced258fc4489cd1f69","title":"Rank-sum test with ties: z is centred on the wrong null mean · case 01","variant":1,"variant_policy":"Five numbered records share a model and may reuse boundary fixtures.","verified":true,"visibility":"public","verification":{"attempt":{"elapsed_ms":41.374,"exit_code":1,"observations":[{"actual":[8.0,2.318694],"check":"ties get average ranks","expected":[8.0,1.623086],"passed":false},{"actual":[3.5,1.224745],"check":"two-way tie across arms","expected":[3.5,1.224745],"passed":true},{"actual":[3.0,1.0],"check":"three-way tie with treatment","expected":[3.0,1.0],"passed":true},{"actual":[9.0,1.944544],"check":"no ties","expected":[9.0,1.06066],"passed":false},{"actual":[6.0,2.020726],"check":"complete separation","expected":[6.0,1.732051],"passed":false},{"actual":[5.0,0.983739],"check":"unequal arm sizes","expected":[5.0,0.491869],"passed":false},{"actual":[3.0,0.894427],"check":"rank sample 1","expected":[3.0,1.341641],"passed":false},{"actual":[18.0,2.330341],"check":"rank sample 2","expected":[18.0,0.559282],"passed":false}],"passed":false,"stderr":"","stdout":"{\"observations\": [{\"check\": \"ties get average ranks\", \"actual\": [8.0, 2.318694], \"expected\": [8.0, 1.623086], \"passed\": false}, {\"check\": \"two-way tie across arms\", \"actual\": [3.5, 1.224745], \"expected\": [3.5, 1.224745], \"passed\": true}, {\"check\": \"three-way tie with treatment\", \"actual\": [3.0, 1.0], \"expected\": [3.0, 1.0], \"passed\": true}, {\"check\": \"no ties\", \"actual\": [9.0, 1.944544], \"expected\": [9.0, 1.06066], \"passed\": false}, {\"check\": \"complete separation\", \"actual\": [6.0, 2.020726], \"expected\": [6.0, 1.732051], \"passed\": false}, {\"check\": \"unequal arm sizes\", \"actual\": [5.0, 0.983739], \"expected\": [5.0, 0.491869], \"passed\": false}, {\"check\": \"rank sample 1\", \"actual\": [3.0, 0.894427], \"expected\": [3.0, 1.341641], \"passed\": false}, {\"check\": \"rank sample 2\", \"actual\": [18.0, 2.330341], \"expected\": [18.0, 0.559282], \"passed\": false}], \"passed\": false}\n"},"broken":{"elapsed_ms":38.068,"exit_code":1,"observations":[{"actual":[8.0,-0.463739],"check":"ties get average ranks","expected":[8.0,1.623086],"passed":false},{"actual":[3.5,-0.408248],"check":"two-way tie across arms","expected":[3.5,1.224745],"passed":false},{"actual":[3.0,-1.0],"check":"three-way tie with treatment","expected":[3.0,1.0],"passed":false},{"actual":[9.0,-1.06066],"check":"no ties","expected":[9.0,1.06066],"passed":false},{"actual":[6.0,0.0],"check":"complete separation","expected":[6.0,1.732051],"passed":false},{"actual":[5.0,-1.475608],"check":"unequal arm sizes","expected":[5.0,0.491869],"passed":false},{"actual":[3.0,0.0],"check":"rank sample 1","expected":[3.0,1.341641],"passed":false},{"actual":[18.0,-2.237127],"check":"rank sample 2","expected":[18.0,0.559282],"passed":false}],"passed":false,"stderr":"","stdout":"{\"observations\": [{\"check\": \"ties get average ranks\", \"actual\": [8.0, -0.463739], \"expected\": [8.0, 1.623086], \"passed\": false}, {\"check\": \"two-way tie across arms\", \"actual\": [3.5, -0.408248], \"expected\": [3.5, 1.224745], \"passed\": false}, {\"check\": \"three-way tie with treatment\", \"actual\": [3.0, -1.0], \"expected\": [3.0, 1.0], \"passed\": false}, {\"check\": \"no ties\", \"actual\": [9.0, -1.06066], \"expected\": [9.0, 1.06066], \"passed\": false}, {\"check\": \"complete separation\", \"actual\": [6.0, 0.0], \"expected\": [6.0, 1.732051], \"passed\": false}, {\"check\": \"unequal arm sizes\", \"actual\": [5.0, -1.475608], \"expected\": [5.0, 0.491869], \"passed\": false}, {\"check\": \"rank sample 1\", \"actual\": [3.0, 0.0], \"expected\": [3.0, 1.341641], \"passed\": false}, {\"check\": \"rank sample 2\", \"actual\": [18.0, -2.237127], \"expected\": [18.0, 0.559282], \"passed\": false}], \"passed\": false}\n"}},"member_only":{"stages":["fixed"],"fields":["implementations.fixed","verification.fixed","harness","repair"],"note":"The verified repair, its recorded checks, the repair description, and the scoring harness are available to members."}}