FA-47356 / Delimited text / Open access
A user column occupies the overflow namespace · case 01
A structured table violates the declared record or column contract.
ROOT CAUSE
Reserved metadata names are admitted as data columns.
VERIFIED REPAIR
Preserve the named invariant at the faulty decision: if '_extra' in headers: return None
Unsuccessful approach: Rejecting only an all-metadata header does not protect mixed tables.
Case contract
Bind a comma-split header and data row. Header keys are ASCII case insensitive and stripped. Reject empty or duplicate canonical names, reserved _extra, and unequal width. Preserve data cell whitespace.
Why this case matters
Delimited interchange needs explicit framing, schema and field semantics at ingestion and emission boundaries.
1 / The failure
Exit 1"""Failure Map reference implementation. Python standard library only."""
import json
def _vary(value):
if value == '@END': return 3 + 4*N
if isinstance(value, str): return value.replace('@', 'cell' * N)
if isinstance(value, list): return [_vary(x) for x in value]
if isinstance(value, dict): return {_vary(k): _vary(v) for k,v in value.items()}
return value
N = 1
observations = []
def solve(data):
headers = [h.strip().lower() for h in data[0].split(',')]
values = data[1].split(',')
if any(not h for h in headers): return None
if len(set(headers)) != len(headers): return None
if False: return None
if len(values) != len(headers): return None
return dict(zip(headers, values))
def check(label, actual, expected):
observations.append({"check": label, "actual": actual, "expected": expected, "passed": actual == expected})
check('canonical names', solve(_vary([' A ,B', '@, y '])), _vary({'a': '@', 'b': ' y '}))
check('duplicate canonical', solve(_vary(['A,a', 'x,y'])), _vary(None))
check('empty header', solve(_vary(['a,', 'x,y'])), _vary(None))
check('reserved header', solve(_vary(['_extra,b', 'x,y'])), _vary(None))
check('short row', solve(_vary(['a,b', '@'])), _vary(None))
check('long row', solve(_vary(['a', 'x,y'])), _vary(None))
check('normal', solve(_vary(['a,b', 'x,y'])), _vary({'a': 'x', 'b': 'y'}))
print(json.dumps({"observations": observations, "passed": all(x["passed"] for x in observations)}, ensure_ascii=False))
raise SystemExit(0 if all(x["passed"] for x in observations) else 1)
| Boundary fixture | Actual | Expected | Outcome |
|---|---|---|---|
| canonical names | {'a': 'cell', 'b': ' y '} | {'a': 'cell', 'b': ' y '} | Passed |
| duplicate canonical | None | None | Passed |
| empty header | None | None | Passed |
| reserved header | {'_extra': 'x', 'b': 'y'} | None | Failed |
| short row | None | None | Passed |
| long row | None | None | Passed |
| normal | {'a': 'x', 'b': 'y'} | {'a': 'x', 'b': 'y'} | Passed |
SHA-256 / a951f2e50d8fc91762172303074fa872b49ad6d48a9eaaa6c5bd4f2de19713f0
2 / The unsuccessful fix
Exit 1"""Failure Map reference implementation. Python standard library only."""
import json
def _vary(value):
if value == '@END': return 3 + 4*N
if isinstance(value, str): return value.replace('@', 'cell' * N)
if isinstance(value, list): return [_vary(x) for x in value]
if isinstance(value, dict): return {_vary(k): _vary(v) for k,v in value.items()}
return value
N = 1
observations = []
def solve(data):
headers = [h.strip().lower() for h in data[0].split(',')]
values = data[1].split(',')
if any(not h for h in headers): return None
if len(set(headers)) != len(headers): return None
if headers[0] == '_extra' and len(headers) == 1: return None
if len(values) != len(headers): return None
return dict(zip(headers, values))
def check(label, actual, expected):
observations.append({"check": label, "actual": actual, "expected": expected, "passed": actual == expected})
check('canonical names', solve(_vary([' A ,B', '@, y '])), _vary({'a': '@', 'b': ' y '}))
check('duplicate canonical', solve(_vary(['A,a', 'x,y'])), _vary(None))
check('empty header', solve(_vary(['a,', 'x,y'])), _vary(None))
check('reserved header', solve(_vary(['_extra,b', 'x,y'])), _vary(None))
check('short row', solve(_vary(['a,b', '@'])), _vary(None))
check('long row', solve(_vary(['a', 'x,y'])), _vary(None))
check('normal', solve(_vary(['a,b', 'x,y'])), _vary({'a': 'x', 'b': 'y'}))
print(json.dumps({"observations": observations, "passed": all(x["passed"] for x in observations)}, ensure_ascii=False))
raise SystemExit(0 if all(x["passed"] for x in observations) else 1)
| Boundary fixture | Actual | Expected | Outcome |
|---|---|---|---|
| canonical names | {'a': 'cell', 'b': ' y '} | {'a': 'cell', 'b': ' y '} | Passed |
| duplicate canonical | None | None | Passed |
| empty header | None | None | Passed |
| reserved header | {'_extra': 'x', 'b': 'y'} | None | Failed |
| short row | None | None | Passed |
| long row | None | None | Passed |
| normal | {'a': 'x', 'b': 'y'} | {'a': 'x', 'b': 'y'} | Passed |
SHA-256 / 8f63f2620ad19513ef0b9eeca603f3f781dbaf71664f2643ea3fce2bc7375bb4
3 / The verified repair
Exit 0"""Failure Map reference implementation. Python standard library only."""
import json
def _vary(value):
if value == '@END': return 3 + 4*N
if isinstance(value, str): return value.replace('@', 'cell' * N)
if isinstance(value, list): return [_vary(x) for x in value]
if isinstance(value, dict): return {_vary(k): _vary(v) for k,v in value.items()}
return value
N = 1
observations = []
def solve(data):
headers = [h.strip().lower() for h in data[0].split(',')]
values = data[1].split(',')
if any(not h for h in headers): return None
if len(set(headers)) != len(headers): return None
if '_extra' in headers: return None
if len(values) != len(headers): return None
return dict(zip(headers, values))
def check(label, actual, expected):
observations.append({"check": label, "actual": actual, "expected": expected, "passed": actual == expected})
check('canonical names', solve(_vary([' A ,B', '@, y '])), _vary({'a': '@', 'b': ' y '}))
check('duplicate canonical', solve(_vary(['A,a', 'x,y'])), _vary(None))
check('empty header', solve(_vary(['a,', 'x,y'])), _vary(None))
check('reserved header', solve(_vary(['_extra,b', 'x,y'])), _vary(None))
check('short row', solve(_vary(['a,b', '@'])), _vary(None))
check('long row', solve(_vary(['a', 'x,y'])), _vary(None))
check('normal', solve(_vary(['a,b', 'x,y'])), _vary({'a': 'x', 'b': 'y'}))
print(json.dumps({"observations": observations, "passed": all(x["passed"] for x in observations)}, ensure_ascii=False))
raise SystemExit(0 if all(x["passed"] for x in observations) else 1)
| Boundary fixture | Actual | Expected | Outcome |
|---|---|---|---|
| canonical names | {'a': 'cell', 'b': ' y '} | {'a': 'cell', 'b': ' y '} | Passed |
| duplicate canonical | None | None | Passed |
| empty header | None | None | Passed |
| reserved header | None | None | Passed |
| short row | None | None | Passed |
| long row | None | None | Passed |
| normal | {'a': 'x', 'b': 'y'} | {'a': 'x', 'b': 'y'} | Passed |
SHA-256 / 50216bd5abca5160fea3b9512c15cdee50dba035faedc7567df31b012c704763
Verification & scope
Deterministic bounded in-memory model. No claim of complete CSV or external format conformance. This reproducer isolates one failure mechanism. Results cover the supplied fixtures. Variants within a family share a test contract and should remain grouped when constructing evaluation splits. Related mechanisms with a shared evaluation_group must also remain together; these controlled models are not independent production incidents.
Observations recorded using Python 3.12.14 at 2026-09-29T14:44:40.724394+00:00.
Case digest / c78e171771be6e36fee0c8ea8b2c9baf93fe388c92529dd952cb7673095ee6bd