{"abstract":"Whitespace tokenization creates empty tokens.","category":"Text processing","checks":4,"contract":"Split on runs of Unicode whitespace with no leading/trailing empty fields.","evaluation_group":"model-6ff376048de9f2e8","failed_approach":"The attempted repair handles the primary example but still violates a separate boundary of the same contract.","family":"xp-split-whitespace-runs","id":"FA-2996","implementations":{"attempt":{"sha256":"e0a9fcce0f12d766cb852019b29934ad1033a8d0a868c147cc21a528f24078b5","source":"\"\"\"Failure Map reference implementation. Python standard library only.\"\"\"\nimport json\nimport re, unicodedata, json, csv, io, html, base64, binascii, codecs, struct\nfrom urllib.parse import quote, unquote, unquote_to_bytes, urlencode, parse_qsl, urlsplit, urlunsplit\nfrom email.header import decode_header, make_header\nfrom email.utils import getaddresses\nfrom xml.etree import ElementTree as ET\nimport shlex, string, textwrap\n\nN = 1\nobservations = []\ndef solve(x):\n    try:\n        return re.split(r'\\s+',x.strip())\n    except Exception as exc:\n        return {\"error\": type(exc).__name__}\ndef check(label, actual, expected):\n    observations.append({\"check\": label, \"actual\": actual, \"expected\": expected, \"passed\": actual == expected})\ncheck('boundary 1', solve(' a  b '), ['a', 'b'])\ncheck('boundary 2', solve('\\tA\\xa0B'), ['A', 'B'])\ncheck('boundary 3', solve(''), [])\ncheck('boundary 4', solve('  '), [])\nprint(json.dumps({\"observations\": observations, \"passed\": all(x[\"passed\"] for x in observations)}, ensure_ascii=False))\nraise SystemExit(0 if all(x[\"passed\"] for x in observations) else 1)\n"},"broken":{"sha256":"a9e35113a056864e6464c8a4b2a7499f214c43b0a4849813fff8ccf1fc76f4e8","source":"\"\"\"Failure Map reference implementation. Python standard library only.\"\"\"\nimport json\nimport re, unicodedata, json, csv, io, html, base64, binascii, codecs, struct\nfrom urllib.parse import quote, unquote, unquote_to_bytes, urlencode, parse_qsl, urlsplit, urlunsplit\nfrom email.header import decode_header, make_header\nfrom email.utils import getaddresses\nfrom xml.etree import ElementTree as ET\nimport shlex, string, textwrap\n\nN = 1\nobservations = []\ndef solve(x):\n    try:\n        return x.split(' ')\n    except Exception as exc:\n        return {\"error\": type(exc).__name__}\ndef check(label, actual, expected):\n    observations.append({\"check\": label, \"actual\": actual, \"expected\": expected, \"passed\": actual == expected})\ncheck('boundary 1', solve(' a  b '), ['a', 'b'])\ncheck('boundary 2', solve('\\tA\\xa0B'), ['A', 'B'])\ncheck('boundary 3', solve(''), [])\ncheck('boundary 4', solve('  '), [])\nprint(json.dumps({\"observations\": observations, \"passed\": all(x[\"passed\"] for x in observations)}, ensure_ascii=False))\nraise SystemExit(0 if all(x[\"passed\"] for x in observations) else 1)\n"},"fixed":{"sha256":"b565c6c6b3cbf4442b5451686ea29952bf50af3791a991877181a1cd1ff3de7d","source":"\"\"\"Failure Map reference implementation. Python standard library only.\"\"\"\nimport json\nimport re, unicodedata, json, csv, io, html, base64, binascii, codecs, struct\nfrom urllib.parse import quote, unquote, unquote_to_bytes, urlencode, parse_qsl, urlsplit, urlunsplit\nfrom email.header import decode_header, make_header\nfrom email.utils import getaddresses\nfrom xml.etree import ElementTree as ET\nimport shlex, string, textwrap\n\nN = 1\nobservations = []\ndef solve(x):\n    try:\n        return x.split()\n    except Exception as exc:\n        return {\"error\": type(exc).__name__}\ndef check(label, actual, expected):\n    observations.append({\"check\": label, \"actual\": actual, \"expected\": expected, \"passed\": actual == expected})\ncheck('boundary 1', solve(' a  b '), ['a', 'b'])\ncheck('boundary 2', solve('\\tA\\xa0B'), ['A', 'B'])\ncheck('boundary 3', solve(''), [])\ncheck('boundary 4', solve('  '), [])\nprint(json.dumps({\"observations\": observations, \"passed\": all(x[\"passed\"] for x in observations)}, ensure_ascii=False))\nraise SystemExit(0 if all(x[\"passed\"] for x in observations) else 1)\n"}},"limitations":" This reproducer isolates one failure mechanism. Results cover the supplied fixtures. Variants within a family share a test contract and should remain grouped when constructing evaluation splits. Related mechanisms with a shared evaluation_group must also remain together; these controlled models are not independent production incidents.","method":"Deterministic executable model with adversarial boundary fixtures.","provenance":{"created_by":"Failure Map","dependencies":"Python standard library","family":"xp-split-whitespace-runs","generated_at":"2026-09-29T14:37:19.908125+00:00","license":"CC0-1.0","python":"3.12.14","seed":1,"split":"open-access"},"relevance":"A local executable model for consumers of structured text; oracle values are authored literals, not outputs copied from the repaired implementation.","repair":"Implement the complete stated contract, including the boundary fixtures: Split on runs of Unicode whitespace with no leading/trailing empty fields.","root_cause":"The implementation applies an operation whose text or grammar semantics violate this contract: Split on runs of Unicode whitespace with no leading/trailing empty fields.","sha256":"404cf0494b91f7856f16fe9b56bd9a5d3e965641e1f42dc41c3d2a4e6e922965","title":"Whitespace tokenization creates empty tokens · case 01","variant":1,"variant_policy":"Five numbered records share a model and may reuse boundary fixtures.","verification":{"attempt":{"elapsed_ms":69.175,"exit_code":1,"observations":[{"actual":["a","b"],"check":"boundary 1","expected":["a","b"],"passed":true},{"actual":["A","B"],"check":"boundary 2","expected":["A","B"],"passed":true},{"actual":[""],"check":"boundary 3","expected":[],"passed":false},{"actual":[""],"check":"boundary 4","expected":[],"passed":false}],"passed":false,"stderr":"","stdout":"{\"observations\": [{\"check\": \"boundary 1\", \"actual\": [\"a\", \"b\"], \"expected\": [\"a\", \"b\"], \"passed\": true}, {\"check\": \"boundary 2\", \"actual\": [\"A\", \"B\"], \"expected\": [\"A\", \"B\"], \"passed\": true}, {\"check\": \"boundary 3\", \"actual\": [\"\"], \"expected\": [], \"passed\": false}, {\"check\": \"boundary 4\", \"actual\": [\"\"], \"expected\": [], \"passed\": false}], \"passed\": false}\n"},"broken":{"elapsed_ms":59.99,"exit_code":1,"observations":[{"actual":["","a","","b",""],"check":"boundary 1","expected":["a","b"],"passed":false},{"actual":["\tA B"],"check":"boundary 2","expected":["A","B"],"passed":false},{"actual":[""],"check":"boundary 3","expected":[],"passed":false},{"actual":["","",""],"check":"boundary 4","expected":[],"passed":false}],"passed":false,"stderr":"","stdout":"{\"observations\": [{\"check\": \"boundary 1\", \"actual\": [\"\", \"a\", \"\", \"b\", \"\"], \"expected\": [\"a\", \"b\"], \"passed\": false}, {\"check\": \"boundary 2\", \"actual\": [\"\\tA B\"], \"expected\": [\"A\", \"B\"], \"passed\": false}, {\"check\": \"boundary 3\", \"actual\": [\"\"], \"expected\": [], \"passed\": false}, {\"check\": \"boundary 4\", \"actual\": [\"\", \"\", \"\"], \"expected\": [], \"passed\": false}], \"passed\": false}\n"},"fixed":{"elapsed_ms":91.499,"exit_code":0,"observations":[{"actual":["a","b"],"check":"boundary 1","expected":["a","b"],"passed":true},{"actual":["A","B"],"check":"boundary 2","expected":["A","B"],"passed":true},{"actual":[],"check":"boundary 3","expected":[],"passed":true},{"actual":[],"check":"boundary 4","expected":[],"passed":true}],"passed":true,"stderr":"","stdout":"{\"observations\": [{\"check\": \"boundary 1\", \"actual\": [\"a\", \"b\"], \"expected\": [\"a\", \"b\"], \"passed\": true}, {\"check\": \"boundary 2\", \"actual\": [\"A\", \"B\"], \"expected\": [\"A\", \"B\"], \"passed\": true}, {\"check\": \"boundary 3\", \"actual\": [], \"expected\": [], \"passed\": true}, {\"check\": \"boundary 4\", \"actual\": [], \"expected\": [], \"passed\": true}], \"passed\": true}\n"}},"verified":true,"visibility":"public"}