#!/usr/bin/env python3 """Fails the build when a translated body drops a page citation the original carries. Why this exists: the site promises, in both languages, that "every claim carries a book, an edition and a page" — /standards/ says it, the rubric says it, and StudyPage prints it under every study in Urdu as well as English. A translation is where that promise quietly breaks. Prose survives translation because a missing sentence is obvious; a citation does not, because "(PCTB 10 pp.101ff; KP 10 pp.121ff; STBB 10 pp.130ff)" reads as apparatus rather than as content, and the Urdu sentence still makes sense without it. The reader who loses it is exactly the reader least able to go and check the English. This ran for the first time against the twenty translated studies and found 56 such drops across eight files — a whole quoted passage and its two citations gone from the 1971 finding in track-pakistan-studies, the entire chapter-provenance clause gone from the KP class 9 record, fourteen page refs gone from lens-patriotism's board profiles. What it compares: page numbers, per heading-delimited block, as multisets. Block alignment rather than whole-file matters — a citation that migrates from one section to another is also a defect, and the translated bodies keep the English heading structure exactly. Only the FIRST number of each citation is counted, in both languages ("pp.16 and 26" and "صفحات 16 اور 26" both count as 16). The comparison stays symmetric that way, and a translator who keeps the anchor almost never drops its tail. What it does not check: that the page is the RIGHT page. A citation moved from p.55 to p.56 in translation reads as one drop and one addition; the drop is reported, which is enough to put a human on it. Usage: python3 scripts/check_citations.py Exit 1 on any dropped citation or any structural divergence. """ from __future__ import annotations import json import pathlib import re import sys from collections import Counter ROOT = pathlib.Path(__file__).resolve().parent.parent CONTENT = ROOT / "site" / "src" / "content" # The long-form pairs: one Markdown body against its translation. PAIRS = [("studies", "studies-ur"), ("articles", "articles-ur")] # The same check, for prose that lives in the dataset rather than in a page body. A book's # analytical note and a comparison's commentary carry page citations exactly as a study does, # and lose them in translation exactly as easily. Each entry is (English file, Urdu file, how # to walk it), where the walk yields (label, english, urdu) triples. JSON_PAIRS = [ ("books.json", None, "books"), ("comparisons.json", None, "comparisons"), ("summaries.json", None, "summaries"), ("findings.json", None, "findings"), ("library.json", None, "library"), ] # English citation forms in use across the corpus: p.87, pp.55–57, p.~1174 (a character # offset into an extract, used where a scanned book has no printed folio), pp.62ff, and # PAGE 82 (the extraction pipeline's own marker, quoted verbatim in evidence tables). # The lookbehind matters: without it a sentence ending in a word like "write-up." or # "group." followed by a numbered list item reads as the citation "p. 11". EN_CITE = re.compile(r"PAGE\s*(\d+)|(? list[tuple[str, str]]: """Split a body into (heading, text) at every heading.""" out: list[tuple[str, str]] = [] head, buf = "(front matter and preamble)", [] for line in text.splitlines(): if HEADING.match(line): out.append((head, "\n".join(buf))) head, buf = line.strip(), [] else: buf.append(line) out.append((head, "\n".join(buf))) return out # A citation rarely stops at one page. "pp.20/52/65", "p.133, 143" and "pp.67–83 and # pp.113–131" all continue past the anchor, and their Urdu counterparts continue the same # way but usually drop the repeated word ("صفحات 67–83 اور 113–131"). Counting the # continuation in both languages is what keeps the comparison symmetric — otherwise every # such citation reads as a drop. MORE = re.compile( r"\s*(?:[–-]\s*\d+)?\s*(?:[,،؛;/&]|اور|and)\s*(?:تقریباً\s*)?~?(\d+)" ) def cites(pattern: re.Pattern[str], text: str) -> Counter[str]: found: Counter[str] = Counter() for m in pattern.finditer(text): for g in m.groups(): if g: found[g] += 1 pos = m.end() while (nxt := MORE.match(text, pos)) is not None: found[nxt.group(1)] += 1 pos = nxt.end() return found def walk(kind: str, data): """Yield (label, english, urdu) for every translated prose field of one dataset.""" if kind == "books": for b in data: if b.get("note") and (b.get("ur") or {}).get("note"): yield f"books.json · {b['id']} · note", b["note"], b["ur"]["note"] elif kind == "comparisons": for key, v in data.items(): if v.get("commentary") and (v.get("ur") or {}).get("commentary"): yield f"comparisons.json · {key}", v["commentary"], v["ur"]["commentary"] elif kind == "summaries": ur = data.get("ur") or {} for slug, en in (data.get("subjects") or {}).items(): u = (ur.get("subjects") or {}).get(slug) if u: yield f"summaries.json · {slug}", en, u if data.get("overall") and ur.get("overall"): yield "summaries.json · overall", data["overall"], ur["overall"] elif kind == "findings": for f in data: for field in ("claim", "walk", "note", "counter", "citeDetail"): u = (f.get("ur") or {}).get(field) if f.get(field) and u: yield f"findings.json · {f['id']} · {field}", f[field], u elif kind == "library": for w in data: for field in ("note", "bearing"): u = (w.get("ur") or {}).get(field) if w.get(field) and u: yield f"library.json · {w['id']} · {field}", w[field], u def main() -> int: problems: list[str] = [] checked = kept = 0 for src_dir, tr_dir in PAIRS: src, tr = CONTENT / src_dir, CONTENT / tr_dir if not src.is_dir() or not tr.is_dir(): continue for en_path in sorted(src.glob("*.md")): ur_path = tr / en_path.name if not ur_path.exists(): continue # An untranslated body is a known gap, not a defect. checked += 1 en_blocks = blocks(en_path.read_text(encoding="utf-8")) ur_blocks = blocks(ur_path.read_text(encoding="utf-8")) if len(en_blocks) != len(ur_blocks): problems.append( f" {tr_dir}/{en_path.name}\n" f" section count differs: {len(en_blocks)} in English, " f"{len(ur_blocks)} translated — the bodies cannot be compared " f"section by section until the headings match." ) continue for (en_head, en_text), (_, ur_text) in zip(en_blocks, ur_blocks): found = cites(EN_CITE, en_text) kept += sum(found.values()) dropped = found - cites(UR_CITE, ur_text) if dropped: pages = ", ".join( f"p.{n}" + (f" ×{c}" if c > 1 else "") for n, c in sorted(dropped.items(), key=lambda kv: int(kv[0])) ) problems.append( f" {tr_dir}/{en_path.name}\n" f" {en_head[:78]}\n" f" dropped: {pages}" ) # The dataset prose, compared whole rather than by section: these are single fields, so # there is nothing to align — a citation is either carried across or it is not. for name, _, kind in JSON_PAIRS: path = ROOT / "analysis" / name if not path.exists(): continue data = json.loads(path.read_text(encoding="utf-8")) for label, en_text, ur_text in walk(kind, data): checked += 1 found = cites(EN_CITE, en_text) kept += sum(found.values()) dropped = found - cites(UR_CITE, ur_text) if dropped: pages = ", ".join( f"p.{n}" + (f" ×{c}" if c > 1 else "") for n, c in sorted(dropped.items(), key=lambda kv: int(kv[0])) ) problems.append(f" {label}\n dropped: {pages}") if problems: n = len(problems) print(f"check_citations: {n} section(s) lose a citation in translation\n") print("\n\n".join(problems)) print( "\nRestore the page reference in the translation's own source file: a study or\n" "article body under analysis/studies-ur or analysis/articles-ur, a book note in\n" "analysis/book-notes--ur.json, a commentary in cell-commentary-ur.json, a\n" "summary in subject-summaries-ur.json, or the record itself in findings.json or\n" "library.json. The copies under site/src/content and site/src/data are generated\n" "by sync_site.sh and build_dataset.py, so editing those is lost on the next build.\n" "A claim without its page is not a claim this site makes in either language." ) return 1 print(f"check_citations: clean ({kept} citations across {checked} translated bodies)") return 0 if __name__ == "__main__": sys.exit(main())