diff --git a/README.md b/README.md index 9e37d62..470ce59 100644 --- a/README.md +++ b/README.md @@ -206,21 +206,21 @@ This code is Apache 2.0, and our model weights use a modified OpenRAIL-M license # Benchmark table -| **Model** | ArXiv | Old Scans Math | Tables | Old Scans | Headers and Footers | Multi column | Long tiny text | Base | Overall | Source | -|:--------------------------|:--------:|:--------------:|:--------:|:---------:|:-------------------:|:------------:|:--------------:|:----:|:--------------:|:------:| -| Datalab API | **90.4** | **90.2** | **90.7** | **54.6** | 91.6 | 83.7 | **92.3** | **99.9** | **86.7 ± 0.8** | Own benchmarks | -| Chandra 2 | 90.2 | 89.3 | 89.9 | 49.8 | 92.5 | 83.5 | 92.1 | 99.6 | 85.9 ± 0.8 | Own benchmarks | -| dots.ocr 1.5 | 85.9 | 85.5 | **90.7** | 48.2 | 94.0 | **85.3** | 81.6 | 99.7 | 83.9 | dots.ocr repo | -| Chandra 1 | 82.2 | 80.3 | 88.0 | 50.4 | 90.8 | 81.2 | **92.3** | **99.9** | 83.1 ± 0.9 | Own benchmarks | -| olmOCR 2 | 83.0 | 82.3 | 84.9 | 47.7 | **96.1** | 83.7 | 81.9 | 99.6 | 82.4 | olmocr repo | -| dots.ocr | 82.1 | 64.2 | 88.3 | 40.9 | 94.1 | 82.4 | 81.2 | 99.5 | 79.1 ± 1.0 | dots.ocr repo | -| olmOCR v0.3.0 | 78.6 | 79.9 | 72.9 | 43.9 | 95.1 | 77.3 | 81.2 | 98.9 | 78.5 ± 1.1 | olmocr repo | -| Datalab Marker v1.10.0 | 83.8 | 69.7 | 74.8 | 32.3 | 86.6 | 79.4 | 85.7 | 99.6 | 76.5 ± 1.0 | Own benchmarks | -| Deepseek OCR | 75.2 | 72.3 | 79.7 | 33.3 | **96.1** | 66.7 | 80.1 | 99.7 | 75.4 ± 1.0 | Own benchmarks | -| Mistral OCR API | 77.2 | 67.5 | 60.6 | 29.3 | 93.6 | 71.3 | 77.1 | 99.4 | 72.0 ± 1.1 | olmocr repo | -| GPT-4o (Anchored) | 53.5 | 74.5 | 70.0 | 40.7 | 93.8 | 69.3 | 60.6 | 96.8 | 69.9 ± 1.1 | olmocr repo | -| Qwen 3 VL 8B | 70.2 | 75.1 | 45.6 | 37.5 | 89.1 | 62.1 | 43.0 | 94.3 | 64.6 ± 1.1 | Own benchmarks | -| Gemini Flash 2 (Anchored) | 54.5 | 56.1 | 72.1 | 34.2 | 64.7 | 61.5 | 71.5 | 95.6 | 63.8 ± 1.2 | olmocr repo | +| **Model** | ArXiv | Old Scans Math | Tables | Old Scans | Headers and Footers | Multi column | Long tiny text | Base | Overall | Source | +|:--------------------------|:--------:|:--------------:|:--------:|:---------:|:-------------------:|:------------:|:--------------:|:--------:|:--------------:|:--------------:| +| Datalab API | **90.4** | **90.2** | 90.7 | **54.6** | 91.6 | 83.7 | 92.3 | **99.9** | **86.7 ± 0.8** | Own benchmarks | +| Chandra 2 | 86.9 | 89.1 | **92.1** | 51.1 | 91.4 | 82.1 | **93.7** | **99.9** | 85.8 ± 0.8 | Own benchmarks | +| dots.ocr 1.5 | 85.9 | 85.5 | 90.7 | 48.2 | 94.0 | **85.3** | 81.6 | 99.7 | 83.9 | dots.ocr repo | +| Chandra 1 | 82.2 | 80.3 | 88.0 | 50.4 | 90.8 | 81.2 | 92.3 | **99.9** | 83.1 ± 0.9 | Own benchmarks | +| olmOCR 2 | 83.0 | 82.3 | 84.9 | 47.7 | **96.1** | 83.7 | 81.9 | 99.6 | 82.4 | olmocr repo | +| dots.ocr | 82.1 | 64.2 | 88.3 | 40.9 | 94.1 | 82.4 | 81.2 | 99.5 | 79.1 ± 1.0 | dots.ocr repo | +| olmOCR v0.3.0 | 78.6 | 79.9 | 72.9 | 43.9 | 95.1 | 77.3 | 81.2 | 98.9 | 78.5 ± 1.1 | olmocr repo | +| Datalab Marker v1.10.0 | 83.8 | 69.7 | 74.8 | 32.3 | 86.6 | 79.4 | 85.7 | 99.6 | 76.5 ± 1.0 | Own benchmarks | +| Deepseek OCR | 75.2 | 72.3 | 79.7 | 33.3 | **96.1** | 66.7 | 80.1 | 99.7 | 75.4 ± 1.0 | Own benchmarks | +| Mistral OCR API | 77.2 | 67.5 | 60.6 | 29.3 | 93.6 | 71.3 | 77.1 | 99.4 | 72.0 ± 1.1 | olmocr repo | +| GPT-4o (Anchored) | 53.5 | 74.5 | 70.0 | 40.7 | 93.8 | 69.3 | 60.6 | 96.8 | 69.9 ± 1.1 | olmocr repo | +| Qwen 3 VL 8B | 70.2 | 75.1 | 45.6 | 37.5 | 89.1 | 62.1 | 43.0 | 94.3 | 64.6 ± 1.1 | Own benchmarks | +| Gemini Flash 2 (Anchored) | 54.5 | 56.1 | 72.1 | 34.2 | 64.7 | 61.5 | 71.5 | 95.6 | 63.8 ± 1.2 | olmocr repo | # Multilingual benchmark table diff --git a/assets/benchmarks/bench.png b/assets/benchmarks/bench.png index 8df0778..20a782a 100644 Binary files a/assets/benchmarks/bench.png and b/assets/benchmarks/bench.png differ diff --git a/chandra/scripts/OLMOCR_BENCH.md b/chandra/scripts/OLMOCR_BENCH.md new file mode 100644 index 0000000..b237b2a --- /dev/null +++ b/chandra/scripts/OLMOCR_BENCH.md @@ -0,0 +1,53 @@ +# Reproducing Chandra OCR 2 on olmOCR-bench + +End-to-end reproduction of Chandra-OCR-2's score on the upstream +[olmOCR-bench](https://github.com/allenai/olmocr) (`allenai/olmOCR-bench`). + +**Reference result:** ~**85.8%** overall. This is slightly below the 85.9% we measured at launch - this is mainly due to some text normalization we used in our launch benchmarking that differed from the standard olmocr benchmark. + +--- + +## 1. Clone + install Chandra + +```bash +git clone https://github.com/datalab-to/chandra.git +cd chandra +pip install -e . # or: pip install chandra-ocr +pip install huggingface_hub unicodeit # for the bench +``` + +## 2. Serve the model with vLLM + +In one terminal (needs a GPU + Docker; see the main README): + +```bash +chandra_vllm # serves datalab-to/chandra-ocr-2 on :8000 +``` + +This uses the repo's serving config. + +## 3. Install the upstream olmOCR bench (for scoring) + +```bash +pip install "olmocr[bench]" +playwright install-deps && playwright install chromium # KaTeX math rendering +``` + +## 4. Run the benchmark + +In a second terminal: + +```bash +python -m chandra.scripts.olmocr_bench --bench-dir ./olmOCR-bench/bench_data +``` + +This downloads olmOCR-bench (first run only), OCRs all ~1,400 pages via the vLLM +server, postprocesses, and prints the upstream score. + +### Useful flags +- `--image-dpi 300` (default) +- `--workers 32` — concurrent vLLM requests. +- `--vllm-api-base http://localhost:8000/v1` — override the server URL. +- `--skip-inference` — reuse candidate markdown already written; only re-score. +- `--skip-scoring` — only produce candidate markdown (prints the olmocr command). +- `--stock` - use stock settings/no chandra format correction \ No newline at end of file diff --git a/chandra/scripts/olmocr_bench.py b/chandra/scripts/olmocr_bench.py new file mode 100644 index 0000000..172ef89 --- /dev/null +++ b/chandra/scripts/olmocr_bench.py @@ -0,0 +1,260 @@ +""" +Reproduce Chandra-OCR-2's score on the upstream olmOCR-bench +(github.com/allenai/olmocr), end to end: + + 1. download the olmOCR-bench dataset (allenai/olmOCR-bench) from HuggingFace, + 2. OCR every bench page with Chandra (this repo) against a running vLLM server, + 3. apply postprocessing to correct Chandra markdown to more standard markdown. + 4. score with `olmocr.bench.benchmark`. + +Reference result: ~85.8% overall. + +Usage: + # 1) in one terminal, serve the model (see OLMOCR_BENCH.md): + chandra_vllm + # 2) in another: + python -m chandra.scripts.olmocr_bench --bench-dir ./olmOCR-bench/bench_data +""" + +import argparse +import glob +import html as _html +import os +import re +import subprocess +import sys + +import unicodeit + +from chandra.input import load_pdf_images +from chandra.model import InferenceManager +from chandra.model.schema import BatchInputItem +from chandra.settings import settings + + +# --------------------------- output-only postprocess --------------------------- +def _latex_to_unicode(s: str) -> str: + """LaTeX -> Unicode for table-cell math, via the `unicodeit` library + (Greek, symbols, ^/_). We pre-strip \\text/\\mathrm and \\frac (which + unicodeit leaves as-is), and post-strip any commands it didn't recognize.""" + s = _html.unescape(s) + s = re.sub(r"\\(?:text|mathrm)\s*\{([^}]*)\}", r"\1", s) + s = re.sub(r"\\frac\s*\{([^{}]*)\}\s*\{([^{}]*)\}", r"\1/\2", s) + try: + s = unicodeit.replace(s) + except Exception: # noqa: BLE001 - never let a stray token kill the doc + pass + s = ( + re.sub(r"\\[a-zA-Z]+", "", s) + .replace("{", "") + .replace("}", "") + .replace("\\", "") + ) + return re.sub(r"\s+", " ", s).strip() + + +# Unicode sub/superscript maps for / digit+operator content. +_SUP = {c: u for c, u in zip("0123456789+-=()n", "⁰¹²³⁴⁵⁶⁷⁸⁹⁺⁻⁼⁽⁾ⁿ")} +_SUB = {c: u for c, u in zip("0123456789+-=()", "₀₁₂₃₄₅₆₇₈₉₊₋₌₍₎")} + + +def _subsup_to_unicode(s: str) -> str: + """2 -> ₂, 2 -> ²""" + s = re.sub( + r"(.*?)", + lambda m: "".join(_SUB.get(c, c) for c in m.group(1)), + s, + flags=re.S | re.I, + ) + s = re.sub( + r"(.*?)", + lambda m: "".join(_SUP.get(c, c) for c in m.group(1)), + s, + flags=re.S | re.I, + ) + return re.sub(r"", "", s) # drop any leftover unmapped tags + + +def _strip_escapes_outside_math(s: str) -> str: + """Drop the stray backslash-escapes chandra markdown conversion leaves behind (\\_ \\* \\$ ...)""" + parts = re.split(r"(\$\$.*?\$\$|\$[^$\n]+\$)", s, flags=re.S) + for i in range(0, len(parts), 2): # even indices are non-math + parts[i] = re.sub(r"\\([_*$%&#.+()\[\]!>~^{}])", r"\1", parts[i]) + return "".join(parts) + + +def postprocess(md: str) -> str: + """Reformat Chandra markdown for olmoCR scoring (output-only). + + * table-cell .. -> Unicode (chandra has latex in tables) + * HTML-unescape inside prose math spans (\\&c. -> \\&c.) + * / -> Unicode (chandra has sup tags in output) + * drop stray escape backslashes outside math spans (introduced by Chandra markdown renderer) + * drop synthesized figure captions (chandra-specific synth captions) + """ + # only tags Chandra leaves are inside tables -> convert to Unicode + s = re.sub( + r"]*>(.*?)", + lambda m: _latex_to_unicode(m.group(1)), + md, + flags=re.S | re.I, + ) + s = re.sub( + r"\$\$(.+?)\$\$", + lambda m: "$$" + _html.unescape(m.group(1)) + "$$", + s, + flags=re.S, + ) + s = re.sub(r"\$([^$\n]+)\$", lambda m: "$" + _html.unescape(m.group(1)) + "$", s) + s = _subsup_to_unicode(s) + s = _strip_escapes_outside_math(s) + # Drop image markdown (Chandra emits synthesized figure descriptions as alt + # text) + s = re.sub(r"!\[[^\]]*\]\([^)]*\)", "", s) + return s + + +def download_bench(bench_dir: str) -> str: + """Ensure the olmOCR-bench data is present; return the bench_data dir.""" + if os.path.isdir(os.path.join(bench_dir, "pdfs")): + return bench_dir + from huggingface_hub import snapshot_download + + print("Downloading allenai/olmOCR-bench from HuggingFace ...") + local = snapshot_download( + repo_id="allenai/olmOCR-bench", + repo_type="dataset", + allow_patterns=["bench_data/**"], + ) + src = os.path.join(local, "bench_data") + os.makedirs(os.path.dirname(bench_dir) or ".", exist_ok=True) + import shutil + + shutil.copytree(src, bench_dir, dirs_exist_ok=True) + return bench_dir + + +def run_inference( + bench_dir, candidate, image_dpi, workers, vllm_api_base, apply_postprocess=True +): + pdf_dir = os.path.join(bench_dir, "pdfs") + pdfs = sorted(glob.glob(os.path.join(pdf_dir, "**", "*.pdf"), recursive=True)) + cand_dir = os.path.join(bench_dir, candidate) + print(f"OCR'ing {len(pdfs)} bench pages with Chandra (dpi={image_dpi}) ...") + mgr = InferenceManager(method="vllm") + + def out_path(pdf): + rel = os.path.relpath(pdf, pdf_dir) + return os.path.join(cand_dir, f"{os.path.splitext(rel)[0]}_pg1_repeat1.md") + + done, buf = 0, [] + + def flush(batch): + nonlocal done + if not batch: + return + items = [BatchInputItem(image=im, prompt_type="ocr_layout") for _, im in batch] + outs = mgr.generate(items, vllm_api_base=vllm_api_base, max_workers=workers) + for (pdf, _im), out in zip(batch, outs): + op = out_path(pdf) + os.makedirs(os.path.dirname(op), exist_ok=True) + md = out.markdown or "" + with open(op, "w", encoding="utf-8") as f: + f.write(postprocess(md) if apply_postprocess else md) + done += len(batch) + print(f" {done}/{len(pdfs)}") + + for pdf in pdfs: + if os.path.exists(out_path(pdf)): # resumable + continue + try: + imgs = load_pdf_images(pdf, page_range=[0], image_dpi=image_dpi) + except Exception as e: + print(f" render failed {pdf}: {e}") + continue + if imgs: + buf.append((pdf, imgs[0])) + if len(buf) >= workers: + flush(buf) + buf = [] + flush(buf) + return cand_dir + + +def main(): + ap = argparse.ArgumentParser( + description="Benchmark Chandra on upstream olmOCR-bench." + ) + ap.add_argument( + "--bench-dir", + default="./olmOCR-bench/bench_data", + help="Path to olmOCR-bench bench_data (downloaded if missing).", + ) + ap.add_argument("--candidate", default="chandra", help="Candidate subdir name.") + ap.add_argument( + "--image-dpi", type=int, default=300, help="Render DPI (300 recommended)." + ) + ap.add_argument("--workers", type=int, default=32, help="Concurrent vLLM requests.") + ap.add_argument( + "--vllm-api-base", + default=settings.VLLM_API_BASE, + help="vLLM OpenAI base URL (default from settings).", + ) + ap.add_argument( + "--skip-inference", + action="store_true", + help="Reuse existing candidate markdown; only score.", + ) + ap.add_argument( + "--skip-scoring", + action="store_true", + help="Only produce candidate markdown; don't run olmocr.bench.", + ) + ap.add_argument( + "--stock", + action="store_true", + help="Stock baseline: render at DPI-192 and apply NO output postprocess", + ) + args = ap.parse_args() + + bench_dir = download_bench(args.bench_dir) + image_dpi = 192 if args.stock else args.image_dpi + if not args.skip_inference: + run_inference( + bench_dir, + args.candidate, + image_dpi, + args.workers, + args.vllm_api_base, + apply_postprocess=not args.stock, + ) + + if args.skip_scoring: + print(f"\nCandidate written to {os.path.join(bench_dir, args.candidate)}") + print( + "Score it with: python -m olmocr.bench.benchmark " + f"--dir {bench_dir} --candidate {args.candidate}" + ) + return + + cmd = [ + sys.executable, + "-m", + "olmocr.bench.benchmark", + "--dir", + bench_dir, + "--candidate", + args.candidate, + ] + print("\nScoring with upstream olmocr.bench:\n " + " ".join(cmd) + "\n") + try: + subprocess.run(cmd, check=True) + except FileNotFoundError: + print( + "olmocr not installed. Install with: pip install olmocr[bench] && playwright install chromium" + ) + sys.exit(1) + + +if __name__ == "__main__": + main()