#!/usr/bin/env python3 """Mark Latin-script blocks in the Urdu edition as left-to-right. The Urdu pages set `dir="rtl"` on , so every block inherits RTL unless it says otherwise. An English paragraph rendered under RTL is not merely right-aligned: the bidirectional algorithm moves its trailing punctuation to the left edge and reorders runs of digits, brackets and citations, which is what makes a fallback English passage read as scrambled rather than merely untranslated. Prose written for this project marks its own Latin runs (``), and the shared components mark whole English bodies. What this pass catches is the remainder — catalogue rows, publisher lines, scale-pole labels, still-English chrome — where the English arrives from data rather than from a translated string, and where marking every site by hand would mean editing hundreds of literals that will change again as translation proceeds. Run after the build. Idempotent: an element that already carries `dir` is left alone. """ import re import sys from html.parser import HTMLParser from pathlib import Path VOID = {'area', 'base', 'br', 'col', 'embed', 'hr', 'img', 'input', 'link', 'meta', 'source', 'track', 'wbr'} # Block containers whose alignment and internal ordering a reader notices. BLOCK = {'p', 'li', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'blockquote', 'td', 'th', 'figcaption', 'dd', 'dt', 'caption', 'summary'} ARABIC = re.compile(r'[؀-ۿݐ-ݿࢠ-ࣿﭐ-﷿ﹰ-]') LATIN = re.compile(r'[A-Za-z]') class Marker(HTMLParser): """Collect the byte offsets of block start-tags that need dir="ltr".""" def __init__(self, line_starts: list[int]) -> None: super().__init__(convert_charrefs=True) self.stack: list[list] = [] # [tag, dir, text, is_block, insert_offset] self.marks: list[tuple[int, str]] = [] # HTMLParser reports a (line, column) position, not an absolute index, so the # index of each line start is precomputed once and added back. self.line_starts = line_starts def _abs(self) -> int: line, col = self.getpos() return self.line_starts[line - 1] + col # -- helpers --------------------------------------------------------- def _inherited(self) -> str: return self.stack[-1][1] if self.stack else 'rtl' def _close(self, index: int) -> None: node = self.stack.pop(index) # Anything still open above the matched tag was never closed; fold its text up. orphans = [self.stack.pop() for _ in range(len(self.stack) - index)] text = node[2] + ''.join(o[2] for o in orphans) if node[3] and node[1] != 'ltr' and self._needs_ltr(text): # A block with no Arabic script at all is English outright, so it is also # tagged lang="en" — that is what gives it the Latin face and tells a screen # reader to switch voice. A mixed block keeps its Urdu language tag and only # has its direction corrected. pure = not ARABIC.search(text) self.marks.append((node[4], ' lang="en" dir="ltr"' if pure else ' dir="ltr"')) elif self.stack: self.stack[-1][2] += text @staticmethod def _needs_ltr(text: str) -> bool: text = text.strip() latin = len(LATIN.findall(text)) arabic = len(ARABIC.findall(text)) if not latin: return False if not arabic: # Pure Latin: even a short catalogue cell reorders its punctuation under RTL. return latin >= 4 # Mixed: only mark when the block is substantially English, so an Urdu paragraph # carrying a board code or an axis name is left as Urdu. return len(text) >= 40 and latin > 30 and latin > arabic * 1.5 # -- parser hooks ---------------------------------------------------- def handle_starttag(self, tag, attrs): if tag in VOID: return a = dict(attrs) own = a.get('dir') if own is None and (a.get('lang') or '').lower().startswith('en'): own = 'ltr' # Offset just past the tag name, where a new attribute can be inserted. # Taken from the source text rather than the normalised tag, which HTMLParser # lower-cases. raw = self.get_starttag_text() or f'<{tag}' name = re.match(r'<([A-Za-z0-9]+)', raw) pos = self._abs() + 1 + len(name.group(1) if name else tag) self.stack.append([tag, own or self._inherited(), '', tag in BLOCK, pos]) def handle_startendtag(self, tag, attrs): return def handle_endtag(self, tag): if tag in VOID: return for i in range(len(self.stack) - 1, -1, -1): if self.stack[i][0] == tag: self._close(i) return def handle_data(self, data): if self.stack: self.stack[-1][2] += data def finish(self) -> list[tuple[int, str]]: while self.stack: self._close(len(self.stack) - 1) return sorted(set(self.marks), reverse=True) MAIN = re.compile(r'', re.S) def process(html: str) -> str: m = MAIN.search(html) if not m: return html region = m.group(0) line_starts, idx = [0], 0 for line in region.split('\n')[:-1]: idx += len(line) + 1 line_starts.append(idx) marker = Marker(line_starts) try: marker.feed(region) offsets = marker.finish() except Exception: return html out = region for off, attr in offsets: out = out[:off] + attr + out[off:] return html[:m.start()] + out + html[m.end():] def main(root: Path) -> int: if not root.is_dir(): print(f'mark_ltr: {root} is not a directory', file=sys.stderr) return 1 changed = added = 0 for path in sorted(root.rglob('*.html')): src = path.read_text(encoding='utf-8') out = process(src) if out != src: path.write_text(out, encoding='utf-8') changed += 1 added += out.count(' dir="ltr"') - src.count(' dir="ltr"') print(f' ltr marks: {added} Latin blocks in {changed} Urdu pages') return 0 if __name__ == '__main__': target = Path(sys.argv[1] if len(sys.argv) > 1 else 'dist/ur') raise SystemExit(main(target))