From f3ab589e3dbded12aaa59b5cc206e12c3ff0aad6 Mon Sep 17 00:00:00 2001 From: Broderick Davidson J Schipp Date: Thu, 30 Jul 2026 13:24:57 -0400 Subject: [PATCH] =?UTF-8?q?preprocessing:=20binarize=20defaults=20OFF=20?= =?UTF-8?q?=E2=80=94=20measured=20accuracy=20win=20on=20three=20collection?= =?UTF-8?q?s=20(v0.8.8)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tesseract 5's LSTM engine thresholds internally and reads grayscale better than anything we hand it pre-binarized. Our adaptive window was flipping background paper grain to ink, which Tesseract emitted as stray punctuation ('.', '|', ':') and glued onto real words ('the.', 'of.'). Measured mean WC, adaptive threshold ON -> OFF: ZeffPapers/MSP00074 (106 pg) 0.828 -> 0.909 review pages 28 -> 6 MSA00005 (11 pg) 0.583 -> 0.781 7 -> 3 MSA00005 (2 pg) 0.453 -> 0.945 2 -> 0 MSA00018 (82 pg) 0.803 -> 0.919 25 -> 6 Token-level diffing confirms the words that disappear are exactly those artifacts while real words are gained, so the lower total word count after the change is noise leaving, not text lost. - PreprocessConfig.do_binarize now False, with the evidence recorded inline. - CLI takes --binarize/--no-binarize; the old --no-binarize spelling still parses so existing scripts keep running. - gui._migrate_settings retires a persisted do_binarize=True once. Saved GUI settings outrank dataclass defaults, so a default flip alone would never have reached anyone who had already run the app. - CLAUDE.md documents the default and, separately, --psm 1 as the fix for collections with rotated scans: it recovers sideways pages and still emits ALTO coordinates in the input frame (verified by overlaying boxes on the untouched scan), but is deliberately NOT the default — it is a wash on unrotated material, costs ~13% wall time, and its OSD manufactures noise on blank pages (a clean empty page became 9 junk words at PC 0.39). Found by the first production pilot, MSA00018 Cornell Bell papers. Outputs produced before this commit were binarized and score lower than a re-run would; re-OCR before upload, same caveat as the pre-v0.8.3 frame fix. Co-Authored-By: Claude Opus 5 --- .gitignore | 6 +- CLAUDE.md | 6 +- src/alto_ocr/__init__.py | 2 +- src/alto_ocr/gui.py | 29 +++++++- src/alto_ocr/main.py | 18 +++-- src/alto_ocr/preprocess.py | 13 +++- tests/test_preprocess_defaults.py | 108 ++++++++++++++++++++++++++++++ 7 files changed, 168 insertions(+), 14 deletions(-) create mode 100644 tests/test_preprocess_defaults.py diff --git a/.gitignore b/.gitignore index f6a3ddd..8f1e5d4 100644 --- a/.gitignore +++ b/.gitignore @@ -16,8 +16,10 @@ _probe_summary.json batch_spotcheck_summary.json temp-ocr*/ -# Local outputs from smoke tests / batch runs — keep schema, not data -output/*/ +# Local outputs from smoke tests / batch runs — keep schema, not data. +# Covers loose run artifacts (summary JSON, batch logs) as well as the +# per-Ref-ID output directories; `output/*/` alone missed the files. +output/* !output/.gitkeep ocr-output-local/ diff --git a/CLAUDE.md b/CLAUDE.md index fe4e56b..5ca97c3 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -58,13 +58,15 @@ Use **OpenCV (`opencv-python-headless`)** as the workhorse — best deskew/thres 3. **Deskew** — estimate skew angle from a Hough-lines or minimum-area-rect approach on a binarized copy; rotate the grayscale image with `cv2.warpAffine` (white border fill). **Frame-changing — force-disabled by both engines' `adjust_preprocess`; see "Coordinate-frame integrity" below.** (The stage remains in `preprocess.py` for any future engine that inverse-transforms its coordinates.) 4. **Denoise** — light bilateral filter (`cv2.bilateralFilter`) on grayscale to smooth scan noise without blurring stroke edges. Skip on bitonal input. 5. **Contrast normalization** — CLAHE (`cv2.createCLAHE`) for faded or unevenly-lit scans. Modest defaults (`clipLimit=2.0`, `tileGridSize=(8,8)`). -6. **Binarization** — adaptive threshold (`cv2.adaptiveThreshold` with Gaussian) or Sauvola if we add `scikit-image`. Adaptive handles page shadows and uneven lighting better than global Otsu. Skip if input is already clean bitonal. +6. **Binarization** — adaptive threshold (`cv2.adaptiveThreshold` with Gaussian). **OFF by default since 2026-07-30 — do not flip it back without re-measuring.** Tesseract 5's LSTM engine does its own thresholding and scores materially higher on grayscale than on anything we hand it pre-binarized. Measured on three unrelated collections (mean WC, ON → OFF): ZeffPapers/`MSP00074` 0.828 → 0.909, `MSA00005` 0.583 → 0.781 and 0.453 → 0.945, `MSA00018` 0.803 → 0.919; pages needing review fell 28 → 6, 9 → 3, 25 → 6. The mechanism is **speckle**: on toned or grainy paper the adaptive window flips background grain to ink, which Tesseract emits as stray punctuation (`.` `|` `:`) and glues onto real words (`the.`, `of.`). Token-level diffing confirmed the words that vanish when it's off are exactly those artifacts, while real words are *gained* — so a lower total word count after turning it off is the noise leaving, not text lost. Opt in per collection with `--binarize` (GUI: the Binarize checkbox) only if triage shows it helping. A persisted GUI preference for the old default is retired once by `gui._migrate_settings`. Locked by `tests/test_preprocess_defaults.py`. 7. **Border crop** — detect and trim solid-black scanning borders that confuse Tesseract's layout analysis (large connected components touching the edge). **Frame-changing — force-disabled by both engines, same as deskew.** **Coordinate-frame integrity (load-bearing).** ALTO box coordinates must land on the **original** image: the public archive draws hit boxes, and the Review tab draws word boxes/snippets, over the untouched preservation scan, scaling by image-px / `Page`-dims. Only the *uniform* DPI upscale cancels in that mapping — deskew (rotation) and border-crop (translation + resize) move pixels relative to the original frame, so both engines force them off in `adjust_preprocess` (`ocr.py`), regardless of CLI/GUI toggles (which can turn stages off, never frame-changers on). Measured on ZeffPapers + MSA00005, 119 pages, 2026-07-30: deskew fired on 36% of pages (median 1.1°, max 4.4°) and put word boxes a full line off their words near page edges; the A/B on the worst pages showed disabling it also **raised** mean WC by +0.034 (Tesseract's baseline fitting handles these angles; the rotation's resampling just blurred glyphs). Border-crop fired on 0/119 (adaptive binarize hollows out solid borders before the crop sees them). If a collection ever appears whose skew genuinely breaks recognition, the fix is inverse-transforming coordinates at emission (the GLMOCR handoff's "v2"), **not** re-enabling deskew. Locked by `tests/test_coordinate_frame.py`. Design rules for the pipeline: -- **Per-stage CLI flags** (`--no-deskew`, `--no-binarize`, `--no-clahe`, `--psm 4`, etc.) so a user can tune per-collection without code changes. Default-on for every stage that's safe; user disables what hurts. +- **Per-stage CLI flags** (`--no-deskew`, `--binarize`/`--no-binarize`, `--no-clahe`, `--psm 4`, etc.) so a user can tune per-collection without code changes. Default-on for every stage that's safe *and measured to help*; user disables what hurts. Binarize is the one stage that defaults **off** (see above); it keeps both spellings so older scripts carrying `--no-binarize` still run. + +**Rotated scans and `--psm` (per-collection, deliberately not a default).** Some collections contain pages scanned sideways — landscape tables in a portrait frame. `--psm 3` (the default) has **no orientation detection**, so Tesseract reads them at 90°/180° and emits fluent-looking garbage at low confidence. `--psm 1` (auto + OSD) fixes them: measured on `MSA00018`, where 16 of 82 pages were rotated (4×90°, 11×180°, 1×270°), the worst table page went 0.28 → 0.73 PC and its transcript from `LUIt-9T9 OSTT-009` to `TABLE I COLLEGE ADMISSIONS DATA FOR THE TEN GROUPS`. Crucially, Tesseract rotates *internally* and still emits ALTO coordinates in the **input image frame** — verified by overlaying the boxes on the untouched scan — so `--psm 1` needs no inverse transform and does not touch the coordinate-frame rule above. **It is not the default** because on the fixtures (no rotated pages) it is a wash (−0.002/+0.000/+0.001 mean WC), costs ~13% wall time, and **manufactures noise on blank pages**: OSD has too little text to orient, guesses, and turned a clean `empty` page (0 words, PC 0.95) into 9 junk words at PC 0.39. Reach for it when triage shows a collection with garbled-but-structured pages; check afterwards that blank pages did not become `scored`. - **Tunable params** for the noisy stages (CLAHE clip, adaptive-threshold block size & C). Reasonable defaults committed; expose flags for the dials. - **Save the preprocessed image** that Tesseract actually saw, optionally (e.g. `--debug-preprocessed `), so accuracy issues are debuggable. Don't write these by default — they double output volume. - **Never mutate the source TIF**. Preprocessing operates on an in-memory copy or a temp file; the original on-disk TIF is never touched. diff --git a/src/alto_ocr/__init__.py b/src/alto_ocr/__init__.py index 755aac5..713e4c8 100644 --- a/src/alto_ocr/__init__.py +++ b/src/alto_ocr/__init__.py @@ -1,3 +1,3 @@ """alto_ocr — TIF page scans → METS/ALTO XML + plain-text transcripts.""" -__version__ = "0.8.7" +__version__ = "0.8.8" diff --git a/src/alto_ocr/gui.py b/src/alto_ocr/gui.py index f27aa1d..dd66912 100644 --- a/src/alto_ocr/gui.py +++ b/src/alto_ocr/gui.py @@ -77,9 +77,36 @@ class JobConfig: glmocr_timeout: int = 120 +# Bumped when a saved preference has to be retired rather than merely +# defaulted. Saved settings normally win over dataclass defaults, so a +# default flip alone would never reach anyone who has run the GUI before. +_BINARIZE_MIGRATION_KEY = "binarize_default_2026_07_30" + + +def _migrate_settings(s: dict) -> dict: + """Retire a persisted ``do_binarize: true``. + + Binarization was default-on until 2026-07-30, when it was measured to cost + accuracy on every collection tested (see PreprocessConfig.do_binarize). A + plain default flip would not reach existing users, whose saved settings + take precedence — so drop the stale preference once and record that we did. + Anyone who deliberately re-checks the box keeps it: the marker means this + runs only the first time. + """ + if s.get(_BINARIZE_MIGRATION_KEY): + return s + if s.get("do_binarize"): + s.pop("do_binarize", None) + logger.info("Binarize is no longer on by default (it lowered OCR accuracy on every " + "collection measured); your saved preference for it has been cleared. " + "Re-check the box if a collection needs it.") + s[_BINARIZE_MIGRATION_KEY] = True + return s + + def _load_settings() -> dict: try: - return json.loads(SETTINGS_PATH.read_text(encoding="utf-8")) + return _migrate_settings(json.loads(SETTINGS_PATH.read_text(encoding="utf-8"))) except (OSError, ValueError): return {} diff --git a/src/alto_ocr/main.py b/src/alto_ocr/main.py index 17f36e5..2e4aec9 100644 --- a/src/alto_ocr/main.py +++ b/src/alto_ocr/main.py @@ -116,14 +116,14 @@ def _build_engine( def _build_preprocess_config( min_dpi: int, no_deskew: bool, no_denoise: bool, no_clahe: bool, - no_binarize: bool, no_border_crop: bool, + binarize: bool, no_border_crop: bool, ) -> PreprocessConfig: return PreprocessConfig( min_dpi=min_dpi, do_deskew=not no_deskew, do_denoise=not no_denoise, do_clahe=not no_clahe, - do_binarize=not no_binarize, + do_binarize=binarize, do_border_crop=not no_border_crop, ) @@ -270,7 +270,11 @@ def _print_triage_hint( NoDeskewOpt = typer.Option(False, "--no-deskew", help="Skip deskew stage.") NoDenoiseOpt = typer.Option(False, "--no-denoise", help="Skip bilateral denoise stage.") NoClaheOpt = typer.Option(False, "--no-clahe", help="Skip CLAHE contrast normalization.") -NoBinarizeOpt = typer.Option(False, "--no-binarize", help="Skip adaptive thresholding (feed grayscale to OCR).") +BinarizeOpt = typer.Option(False, "--binarize/--no-binarize", + help="Apply adaptive thresholding before OCR. OFF by default — Tesseract 5 " + "does its own thresholding and scores higher on grayscale; ours turns " + "paper grain into speckle. Opt in only if a collection measurably " + "benefits (check mean WC with the triage command).") NoBorderCropOpt = typer.Option(False, "--no-border-crop", help="Skip black-border auto-crop.") NoChecksumsOpt = typer.Option(False, "--no-checksums", help="Skip MD5 checksums in METS (faster on large objects).") NoMetsOpt = typer.Option(False, "--no-mets", @@ -310,7 +314,7 @@ def process( no_deskew: bool = NoDeskewOpt, no_denoise: bool = NoDenoiseOpt, no_clahe: bool = NoClaheOpt, - no_binarize: bool = NoBinarizeOpt, + binarize: bool = BinarizeOpt, no_border_crop: bool = NoBorderCropOpt, no_checksums: bool = NoChecksumsOpt, no_mets: bool = NoMetsOpt, @@ -344,7 +348,7 @@ def process( glmocr_num_predict=glmocr_num_predict, ) - cfg = _build_preprocess_config(min_dpi, no_deskew, no_denoise, no_clahe, no_binarize, no_border_crop) + cfg = _build_preprocess_config(min_dpi, no_deskew, no_denoise, no_clahe, binarize, no_border_crop) progress = PageProgressLogger(inputs.page_count) result = process_ref_id( @@ -458,7 +462,7 @@ def batch( no_deskew: bool = NoDeskewOpt, no_denoise: bool = NoDenoiseOpt, no_clahe: bool = NoClaheOpt, - no_binarize: bool = NoBinarizeOpt, + binarize: bool = BinarizeOpt, no_border_crop: bool = NoBorderCropOpt, no_checksums: bool = NoChecksumsOpt, no_mets: bool = NoMetsOpt, @@ -491,7 +495,7 @@ def batch( glmocr_num_predict=glmocr_num_predict, ) - cfg = _build_preprocess_config(min_dpi, no_deskew, no_denoise, no_clahe, no_binarize, no_border_crop) + cfg = _build_preprocess_config(min_dpi, no_deskew, no_denoise, no_clahe, binarize, no_border_crop) results: list[RefIdResult] = [] if workers <= 1: diff --git a/src/alto_ocr/preprocess.py b/src/alto_ocr/preprocess.py index 0371bfb..b40de8e 100644 --- a/src/alto_ocr/preprocess.py +++ b/src/alto_ocr/preprocess.py @@ -50,7 +50,18 @@ class PreprocessConfig: do_deskew: bool = True do_denoise: bool = True do_clahe: bool = True - do_binarize: bool = True + # OFF by default: Tesseract 5's LSTM engine does its own thresholding and + # reads grayscale better than anything we hand it pre-binarized. Measured + # 2026-07-30 across three unrelated collections (mean WC, adaptive-threshold + # ON -> OFF): ZeffPapers/MSP00074 0.828 -> 0.909, MSA00005 0.583 -> 0.781 + # and 0.453 -> 0.945, MSA00018 0.803 -> 0.919. Pages needing review fell + # 28 -> 6, 9 -> 3 and 25 -> 6 respectively. The mechanism is speckle: on + # toned or grainy paper the adaptive window flips background grain to ink, + # which Tesseract emits as stray punctuation ('.', '|', ':') and glues onto + # real words ('the.', 'of.'). Token analysis confirmed the words that + # disappear when it is off are those artifacts, while real words are gained. + # Turn back on per collection with --binarize if one ever benefits. + do_binarize: bool = False do_border_crop: bool = True deskew_min_angle_deg: float = 0.2 # below this, skip rotation diff --git a/tests/test_preprocess_defaults.py b/tests/test_preprocess_defaults.py new file mode 100644 index 0000000..04982a1 --- /dev/null +++ b/tests/test_preprocess_defaults.py @@ -0,0 +1,108 @@ +"""Binarization defaults OFF, and a stale saved GUI preference is retired once. + +Until 2026-07-30 the pipeline handed Tesseract an adaptive-thresholded image. +Measuring three unrelated collections showed that costs accuracy everywhere, +because Tesseract 5's LSTM engine thresholds internally and our adaptive window +turns paper grain into speckle it then reads as punctuation: + + ZeffPapers/MSP00074 mean WC 0.828 -> 0.909 review pages 28 -> 6 + MSA00005 (11 pages) 0.583 -> 0.781 7 -> 3 + MSA00005 (2 pages) 0.453 -> 0.945 2 -> 0 + MSA00018 (82 pages) 0.803 -> 0.919 25 -> 6 + +These tests pin the resulting default so it cannot drift back silently, and +cover the GUI migration — saved settings normally win over dataclass defaults, +so without it nobody who had already run the GUI would ever see the change. + +Run from the repo root (so ``src.alto_ocr`` resolves): + + python -m unittest tests.test_preprocess_defaults +""" + +from __future__ import annotations + +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +import numpy as np # noqa: E402 + +from src.alto_ocr.gui import _BINARIZE_MIGRATION_KEY, _migrate_settings # noqa: E402 +from src.alto_ocr.main import _build_preprocess_config # noqa: E402 +from src.alto_ocr.preprocess import PreprocessConfig # noqa: E402 + + +class TestBinarizeDefault(unittest.TestCase): + def test_config_default_is_off(self): + self.assertFalse(PreprocessConfig().do_binarize) + + def test_other_value_stages_still_default_on(self): + """Only binarize changed — denoise/CLAHE were not part of the finding.""" + cfg = PreprocessConfig() + self.assertTrue(cfg.do_denoise) + self.assertTrue(cfg.do_clahe) + + def test_cli_default_leaves_binarize_off(self): + cfg = _build_preprocess_config(300, False, False, False, False, False) + self.assertFalse(cfg.do_binarize) + + def test_cli_binarize_flag_opts_in(self): + cfg = _build_preprocess_config(300, False, False, False, True, False) + self.assertTrue(cfg.do_binarize) + + +class TestGrayscaleReachesOCR(unittest.TestCase): + """With binarize off the pipeline must hand OCR continuous grayscale, not a + two-valued image — that is the whole point of the default.""" + + def test_default_output_is_not_two_valued(self): + from src.alto_ocr.preprocess import preprocess + import tempfile + + from PIL import Image + + # A gradient with text-like dark marks: binarizing collapses it to {0,255}. + arr = np.tile(np.linspace(90, 210, 400, dtype=np.uint8), (300, 1)) + arr[100:140, 50:350] = 20 + with tempfile.TemporaryDirectory() as td: + src = Path(td) / "page.tif" + Image.fromarray(arr).save(src, dpi=(300, 300)) + + out = preprocess(src, PreprocessConfig()) + self.assertGreater(len(np.unique(out.image)), 2, + "default pipeline handed OCR a binarized image") + self.assertNotIn("binarize", out.stages_applied) + + opted_in = preprocess(src, PreprocessConfig(do_binarize=True)) + self.assertEqual(sorted(np.unique(opted_in.image).tolist()), [0, 255]) + self.assertIn("binarize", opted_in.stages_applied) + + +class TestGuiSettingsMigration(unittest.TestCase): + def test_stale_true_preference_is_dropped_once(self): + s = _migrate_settings({"do_binarize": True, "min_dpi": 300}) + self.assertNotIn("do_binarize", s) + self.assertTrue(s[_BINARIZE_MIGRATION_KEY]) + self.assertEqual(s["min_dpi"], 300, "migration must not disturb other settings") + + def test_deliberate_reopt_in_survives_second_run(self): + """Once migrated, a user who re-checks the box keeps it.""" + s = _migrate_settings({"do_binarize": True}) + s["do_binarize"] = True # user re-checks it in the GUI, settings saved + again = _migrate_settings(s) + self.assertTrue(again["do_binarize"]) + + def test_false_preference_is_left_alone(self): + s = _migrate_settings({"do_binarize": False}) + self.assertFalse(s["do_binarize"]) + + def test_empty_settings_are_safe(self): + s = _migrate_settings({}) + self.assertTrue(s[_BINARIZE_MIGRATION_KEY]) + self.assertNotIn("do_binarize", s) + + +if __name__ == "__main__": + unittest.main()