Repository navigation
Expand file tree
/
Copy patherrata.py
More file actions
239 lines (203 loc) · 9.78 KB
/
Copy patherrata.py
File metadata and controls
239 lines (203 loc) · 9.78 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
#!/usr/bin/env python3
"""Apply errata.json to extracted records, and refuse to let one be forgotten.
python3 errata.py --apply correct the cells, keeping the printed value
python3 errata.py --check exit 1 if a correctable cell was not corrected
python3 errata.py --self-test
--errata PATH and --extracted DIR override the defaults (./errata.json
and ./extracted next to this script), for running against a data
directory that isn't this repo's own root.
Why this is a tool and not a note
---------------------------------
`prec:0000` says the primary source is the truth. `err:0001` is a cell where
the primary source is wrong and a later printing fixes it, so precedence would
faithfully import a value the foundation document got wrong -- and one the
extractor's own invariant already flags. The call was: take the later-printing
correction, record both values as an erratum.
One cell does not need a tool. The FAILURE MODE does. Some tables are not
extracted yet, so this erratum sits inert until a later extraction pass runs
-- and an inert correction that everyone has forgotten about is this project's
coverage trap in its purest form: nothing would ever say the typo shipped. So
`--check` is wired into the export step, which refuses to ship a record an
erratum could have corrected and did not.
Both values survive
-------------------
`value` becomes the correction and `printed_value` keeps what the book actually
says, with the erratum id and its evidence attached to the record. A correction
that erased the printed value would be indistinguishable from a misread.
It refuses rather than guesses
------------------------------
If the extracted cell does not already read exactly what errata.json says was
printed, nothing is written. Either the extractor misread the cell or the
erratum is stale, and both are worth a human's attention -- silently
overwriting whatever was there would destroy the evidence that told you.
"""
from __future__ import annotations
import argparse
import json
import sys
from pathlib import Path
TOOL_ROOT = Path(__file__).resolve().parent
ERRATA = TOOL_ROOT / "errata.json"
EXTRACTED = TOOL_ROOT / "extracted"
def load_errata(path: Path = ERRATA) -> list[dict]:
if not path.exists():
return []
return json.loads(path.read_text(encoding="utf-8")).get("errata", [])
def _targets(rec: dict, e: dict) -> bool:
"""Does this extracted record hold the cell the erratum names?"""
if str((rec.get("source") or {}).get("source", "")) != e["source"]:
return False
if str(rec.get("field", "")) != e["field"]:
return False
key = rec.get("key") or {}
return all(str(key.get(k, "")) == str(v) for k, v in e["key"].items())
def scan(extracted: Path, errata: list[dict]) -> list[dict]:
"""Per erratum: the records it targets, and whether each is already applied."""
out = []
for e in errata:
path = extracted / f"{e['table']}.jsonl"
hits: list[tuple[int, dict]] = []
if path.exists():
for i, line in enumerate(path.open()):
if not line.strip():
continue
rec = json.loads(line)
if _targets(rec, e):
hits.append((i, rec))
out.append({"erratum": e, "path": path, "hits": hits})
return out
def apply(extracted: Path = EXTRACTED, errata: list[dict] | None = None) -> tuple[int, list[str]]:
errata = load_errata() if errata is None else errata
problems: list[str] = []
applied = 0
for job in scan(extracted, errata):
e, path, hits = job["erratum"], job["path"], job["hits"]
if not hits:
continue
lines = path.read_text(encoding="utf-8").splitlines()
for i, rec in hits:
if rec.get("erratum", {}).get("id") == e["id"]:
continue # already applied; idempotent
if str(rec.get("value", "")) != e["printed"]:
problems.append(
f"{e['id']}: {path.name} record {rec.get('id')} reads "
f"{rec.get('value')!r}, but the erratum says the book prints "
f"{e['printed']!r}. NOT corrected -- the extraction and the "
f"erratum disagree about what is on the page."
)
continue
rec["printed_value"] = e["printed"]
rec["value"] = e["corrected"]
rec["erratum"] = {
"id": e["id"],
"printed": e["printed"],
"corrected": e["corrected"],
"corrected_by": e.get("corrected_by"),
"evidence": e.get("evidence"),
}
lines[i] = json.dumps(rec, ensure_ascii=False)
applied += 1
path.write_text("\n".join(lines) + "\n", encoding="utf-8")
return applied, problems
def check(extracted: Path = EXTRACTED, errata: list[dict] | None = None) -> list[str]:
"""Every erratum whose cell EXISTS must have been applied to it."""
errata = load_errata() if errata is None else errata
unapplied: list[str] = []
for job in scan(extracted, errata):
e, hits = job["erratum"], job["hits"]
for _i, rec in hits:
if rec.get("erratum", {}).get("id") != e["id"]:
unapplied.append(
f"{e['id']}: {rec.get('id')} ({e['table']} "
f"{e['key']} {e['field']}) reads {rec.get('value')!r} and "
f"was never corrected to {e['corrected']!r}"
)
return unapplied
def self_test() -> int:
import tempfile
failures = 0
def ck(name: str, cond: bool, detail: str = "") -> None:
nonlocal failures
print(f" {'ok ' if cond else 'FAIL'} {name}" + ("" if cond else f" {detail}"))
failures += not cond
e = {
"id": "err:test", "source": "bk", "table": "t", "page": 1,
"key": {"Category": "Widget", "Row": "7"},
"field": "Value", "printed": "12", "corrected": "15",
}
def rec(value: str, **kw):
return dict(
{"id": "bk:p12:r7:Value", "table": "t",
"key": {"Category": "Widget", "Row": "7"},
"field": "Value", "value": value,
"source": {"source": "bk", "page": 12}}, **kw)
with tempfile.TemporaryDirectory() as t:
d = Path(t)
(d / "t.jsonl").write_text(
json.dumps(rec("12")) + "\n"
+ json.dumps(dict(rec("99"), id="other", key={"Category": "Widget", "Row": "6"})) + "\n",
encoding="utf-8")
ck("an uncorrected cell is reported by --check", len(check(d, [e])) == 1)
n, probs = apply(d, [e])
rows = [json.loads(x) for x in (d / "t.jsonl").read_text().splitlines()]
ck("exactly one cell is corrected", n == 1 and not probs, f"{n} {probs}")
ck("the correction is the value that ships", rows[0]["value"] == "15")
ck("the printed value is KEPT, not erased", rows[0]["printed_value"] == "12")
ck("the record names the erratum", rows[0]["erratum"]["id"] == "err:test")
ck("a neighbouring cell is untouched",
rows[1]["value"] == "99" and "erratum" not in rows[1])
ck("--check passes once applied", check(d, [e]) == [])
ck("applying twice changes nothing", apply(d, [e])[0] == 0)
with tempfile.TemporaryDirectory() as t:
d = Path(t)
(d / "t.jsonl").write_text(json.dumps(rec("9")) + "\n", encoding="utf-8")
n, probs = apply(d, [e])
rows = [json.loads(x) for x in (d / "t.jsonl").read_text().splitlines()]
ck("a cell that does not read what the source claims is REFUSED",
n == 0 and len(probs) == 1 and rows[0]["value"] == "9", f"{n} {probs}")
with tempfile.TemporaryDirectory() as t:
# A table not extracted yet: the erratum must be inert, not an error.
ck("an erratum whose table does not exist yet is silent",
apply(Path(t), [e]) == (0, []) and check(Path(t), [e]) == [])
print("\n" + ("all checks passed" if not failures else f"{failures} FAILED"))
return 1 if failures else 0
def main(argv: list[str] | None = None) -> int:
ap = argparse.ArgumentParser(description=__doc__.splitlines()[0],
formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--apply", action="store_true")
ap.add_argument("--check", action="store_true")
ap.add_argument("--errata", type=Path, default=ERRATA,
help="errata.json path (default: %(default)s)")
ap.add_argument("--extracted", type=Path, default=EXTRACTED,
help="extracted/ directory (default: %(default)s)")
ap.add_argument("--self-test", action="store_true")
a = ap.parse_args(argv)
if a.self_test:
return self_test()
errata = load_errata(a.errata)
jobs = scan(a.extracted, errata)
print(f"{a.errata}: {len(errata)} erratum(s)")
for j in jobs:
e = j["erratum"]
where = f"{len(j['hits'])} matching record(s)" if j["hits"] else "not extracted yet -- inert"
print(f" {e['id']} {e['source']} p{e['page']} {e['table']} {e['key']} {e['field']}"
f" {e['printed']!r} -> {e['corrected']!r} [{where}]")
if a.apply:
n, problems = apply(a.extracted, errata)
print(f"\napplied {n} correction(s)")
for p in problems:
print(f" !! {p}")
return 1 if problems else 0
if a.check:
bad = check(a.extracted, errata)
if bad:
print(f"\n!! {len(bad)} erratum target(s) NOT corrected:")
for b in bad:
print(f" {b}")
print("\n Run `python3 errata.py --apply`. A known-wrong value that")
print(" ships looks exactly like a correct one.")
return 1
print("\nno uncorrected erratum targets")
return 0
if __name__ == "__main__":
sys.exit(main())