Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions examples/doc-field-extraction-omni/.gitignore
Original file line number Diff line number Diff line change
@@ -1,3 +1,5 @@
data/
runs/
__pycache__/
.venv/
.ruff_cache/
301 changes: 239 additions & 62 deletions examples/doc-field-extraction-omni/README.md

Large diffs are not rendered by default.

255 changes: 201 additions & 54 deletions examples/doc-field-extraction-omni/fetch.py
100644 → 100755
Original file line number Diff line number Diff line change
@@ -1,77 +1,224 @@
"""Fetch the immutable, public document-field count evidence; no credentials."""
# ruff: noqa: INP001 - Standalone example scripts are not a Python package.
#!/usr/bin/env python3
"""Download the pinned Omni documents and the published replies of one set, checking every hash.

python3 fetch.py --set confirm
python3 fetch.py --set pilot

Standard library only, except that the pilot needs Pillow for one re-encoded page (see below). No token,
no account, no API key: both datasets are public and are read anonymously at pinned revisions, never `main`.

Everything lands in data/ (or --cache):

data/omni/metadata.jsonl the Omni rows: each document's JSON schema and gold JSON
data/omni/images/ the set's page images, each checked against sets/<set>.image_manifest.json
data/packet/<set>/... the published replies, per-document scores and readable request bodies

run.py and score.py call the same functions, so they also download whatever is missing.

Pilot document omni-407 is a PNG over 5 MB as base64, so the study re-encoded it once to JPEG quality 90 and
sent the re-encoded bytes. Pillow 12.2.0 reproduces those bytes exactly; other Pillow or libjpeg builds may not,
and the fetch stops if the hash differs. The confirmation set has no such page.
"""

from __future__ import annotations

import argparse
import hashlib
import io
import json
import os
import tempfile
import time
import urllib.request
from concurrent.futures import ThreadPoolExecutor
from pathlib import Path
from typing import Any

REVISION = "61e751a5f7f374cacd84ccace621f71e8a498404"
PREFIX = "doc-field-extraction-omni/2026-10-01"
BASE = f"https://huggingface.co/datasets/superlinked/sie-task-evidence/resolve/{REVISION}/{PREFIX}"
MANIFEST_SHA256 = "bc4dbde435a57bb243e664ab15fd626e17a7179a177bdb448221bef257c96f95"
FILES = frozenset({"README.md", "summary.json", "counts.jsonl", "inputs.jsonl"})
HERE = Path(__file__).resolve().parent
DATA = HERE / "data"
PINS = json.loads((HERE / "pins.json").read_text(encoding="utf-8"))
SETS = ("pilot", "confirm")
# A sanity bound on one page image; the largest pinned source is 4.3 MB, and every file is also hash-checked.
MAX_IMAGE_BYTES = 16 * 1024 * 1024
PUBLISHED = {
"pilot": {
"replies": "pilot/grader/A1.jsonl",
"scores": "pilot/grader/omni_scores.json",
"show": "pilot/requests/A1.bodies.show.jsonl",
},
"confirm": {
"replies": "confirm/grader/A1.jsonl",
"scores": "confirm/grader/omni_scores.json",
"show": "confirm/requests/A1.confirm.bodies.show.jsonl",
"bodies_sha256": "confirm/requests/A1.confirm.bodies.jsonl.sha256",
},
}


def http_get(url: str, max_bytes: int | None = None, attempts: int = 5) -> bytes:
"""GET a pinned public file, retrying transient failures."""
request = urllib.request.Request(url, headers={"User-Agent": "sie-doc-field-extraction-omni/1.0"})
for attempt in range(attempts):
try:
with urllib.request.urlopen(request, timeout=120) as response: # pinned HTTPS origins
data = response.read() if max_bytes is None else response.read(max_bytes + 1)
except OSError:
if attempt == attempts - 1:
raise
time.sleep(2 + 3 * attempt)
continue
if max_bytes is not None and len(data) > max_bytes:
raise SystemExit(f"{url}: larger than its pinned size")
return data
raise AssertionError("unreachable")


def sha256(data: bytes) -> str:
return hashlib.sha256(data).hexdigest()


def write_atomic(path: Path, data: bytes) -> None:
"""Write a whole file or nothing, so an interrupted or concurrent run never leaves a truncated cache entry."""
handle, name = tempfile.mkstemp(dir=path.parent, prefix=f".{path.name}.", suffix=".part")
partial = Path(name)
try:
with os.fdopen(handle, "wb") as out:
out.write(data)
partial.replace(path)
except BaseException:
partial.unlink(missing_ok=True)
raise


def read_public_file(name: str, max_bytes: int) -> bytes:
"""Read a size-bounded file from the pinned public HTTPS location."""
with urllib.request.urlopen(f"{BASE}/{name}", timeout=60) as response: # noqa: S310 - Pinned HTTPS origin.
data = response.read(max_bytes + 1)
if len(data) > max_bytes:
raise ValueError(f"Evidence exceeds its declared size: {name}")
def cached_pinned(url: str, path: Path, size: int, digest: str) -> bytes:
"""Return a pinned file, downloading it once into the cache and checking its size and sha256."""
if path.exists():
data = path.read_bytes()
if len(data) == size and sha256(data) == digest:
return data
data = http_get(url, max_bytes=size)
if len(data) != size or sha256(data) != digest:
raise SystemExit(f"{url}: size or sha256 does not match the pin")
path.parent.mkdir(parents=True, exist_ok=True)
write_atomic(path, data)
return data


def load_manifest(data: bytes) -> dict:
"""Authenticate the immutable manifest before accepting its flat file list."""
if hashlib.sha256(data).hexdigest() != MANIFEST_SHA256:
raise ValueError("Evidence manifest hash mismatch")
manifest = json.loads(data)
if set(manifest["files"]) != FILES:
raise ValueError("Evidence manifest has an unexpected file set")
return manifest


def verify_file(name: str, data: bytes, expected: dict) -> None:
"""Require the declared byte size and SHA-256 for an allowed file."""
if name not in FILES:
raise ValueError(f"Unexpected evidence file: {name}")
if len(data) != expected["bytes"]:
raise ValueError(f"Evidence size mismatch: {name}")
if hashlib.sha256(data).hexdigest() != expected["sha256"]:
raise ValueError(f"Evidence hash mismatch: {name}")


def fetch(output: Path) -> None:
"""Download and verify the whole packet before replacing local files."""
manifest_bytes = read_public_file("manifest.json", 64 * 1024)
manifest = load_manifest(manifest_bytes)
output.parent.mkdir(parents=True, exist_ok=True)
# Verify the entire download before replacing any existing local evidence.
with tempfile.TemporaryDirectory(prefix="dfe-evidence-", dir=output.parent) as temporary:
staging = Path(temporary)
for name, expected in manifest["files"].items():
data = read_public_file(name, expected["bytes"])
verify_file(name, data, expected)
(staging / name).write_bytes(data)
(staging / "manifest.json").write_bytes(manifest_bytes)
output.mkdir(parents=True, exist_ok=True)
for name in sorted(FILES | {"manifest.json"}):
(staging / name).replace(output / name)
print(json.dumps({"revision": REVISION, "files": len(FILES) + 1, "verified": True}))
def load_omni_rows(cache: Path) -> dict[str, dict[str, Any]]:
"""The pinned Omni metadata rows, keyed as omni-<id>."""
pin = PINS["omni"]["metadata"]
data = cached_pinned(
f"{PINS['omni']['base_url']}/{pin['path']}", cache / "omni" / pin["path"], pin["bytes"], pin["sha256"]
)
rows = {}
for line in data.decode("utf-8").splitlines():
if line.strip():
row = json.loads(line)
rows[f"omni-{row['id']}"] = row
return rows


def load_set(name: str) -> list[dict[str, Any]]:
"""The published image manifest of a set, verified against its pin, in request order."""
spec = PINS["sets"][name]
data = (HERE / spec["image_manifest"]).read_bytes()
if sha256(data) != spec["image_manifest_sha256"]:
raise SystemExit(f"{spec['image_manifest']}: sha256 does not match the published manifest")
rows = json.loads(data)
for row in rows:
row.setdefault("slot", "study")
study = sorted(row["id"] for row in rows if row["slot"] == "study")
if len(study) != spec["study_documents"] or sha256("\n".join(study).encode()) != spec["study_ids_sha256"]:
raise SystemExit(f"{name}: study id list does not match its pinned hash")
return rows


def study_ids(name: str) -> list[str]:
"""The scored documents of a set; the pilot also has six unscored display rows."""
return [row["id"] for row in load_set(name) if row["slot"] == "study"]


def derive_image(doc_id: str, source: bytes, spec: dict[str, Any]) -> bytes:
"""Re-encode a page the study re-encoded (pilot omni-407 only)."""
try:
from PIL import Image # optional: only the pilot needs it
except ImportError as exc:
raise SystemExit(f"{doc_id} is a re-encoded page; install Pillow ({spec['reproduced_with']} matched)") from exc
out = io.BytesIO()
Image.open(io.BytesIO(source)).convert("RGB").save(out, "JPEG", quality=90)
return out.getvalue()


def prepare_images(set_name: str, cache: Path) -> dict[str, Path]:
"""Download every image of a set once, check it against the published manifest, return id -> file."""
rows = load_set(set_name)
omni = load_omni_rows(cache)
derived = PINS["sets"][set_name]["derived_images"]
folder = cache / "omni" / "images"
folder.mkdir(parents=True, exist_ok=True)

def one(row: dict[str, Any]) -> tuple[str, Path]:
doc_id = row["id"]
name = omni[doc_id]["file_name"]
spec = derived.get(doc_id)
if spec is None and name.split("/")[-1] != row["image_file"]:
raise SystemExit(f"{doc_id}: Omni file {name} is not the manifest's {row['image_file']}")
if spec is not None and name != spec["source_file"]:
raise SystemExit(f"{doc_id}: Omni file {name} is not the pinned source {spec['source_file']}")
source_sha = spec["source_sha256"] if spec else row["image_sha256"]
path = folder / name.split("/")[-1]
data = path.read_bytes() if path.exists() else b""
if sha256(data) != source_sha:
data = http_get(f"{PINS['omni']['base_url']}/{name}", max_bytes=MAX_IMAGE_BYTES)
if sha256(data) != source_sha:
raise SystemExit(f"{doc_id}: downloaded {name} does not match its pinned sha256")
write_atomic(path, data)
if spec is not None:
image = derive_image(doc_id, data, spec)
if sha256(image) != row["image_sha256"]:
raise SystemExit(
f"{doc_id}: the JPEG q90 re-encode gives {sha256(image)}, not the published "
f"{row['image_sha256']}. The study used {spec['reproduced_with']}; other Pillow or libjpeg "
"builds can encode differently."
)
path = cache / "omni" / "derived" / row["image_file"]
path.parent.mkdir(parents=True, exist_ok=True)
write_atomic(path, image)
return doc_id, path

with ThreadPoolExecutor(8) as pool:
return dict(pool.map(one, rows))


def fetch_published(set_name: str, cache: Path) -> dict[str, Path]:
"""The published replies, per-document scores and readable bodies of a set, verified against their pins."""
paths = {}
for role, rel in PUBLISHED[set_name].items():
pin = PINS["packet"]["files"][rel]
path = cache / "packet" / rel
data = cached_pinned(f"{PINS['packet']['base_url']}/{rel}", path, pin["bytes"], pin["sha256"])
if role == "bodies_sha256" and data.decode().split()[0] != PINS["sets"][set_name]["bodies_sha256"]:
raise SystemExit(f"{rel} does not hold the pinned bodies sha256")
paths[role] = path
return paths


def main() -> None:
"""Fetch the pinned packet into the requested local directory."""
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--output", type=Path, default=HERE / "data")
"""Fetch and verify one set."""
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--set", choices=SETS, required=True)
parser.add_argument("--cache", type=Path, default=DATA, help="download folder (default data/)")
args = parser.parse_args()
fetch(args.output)
images = prepare_images(args.set, args.cache)
published = fetch_published(args.set, args.cache)
report = {
"set": args.set,
"omni_revision": PINS["omni"]["revision"],
"evidence_revision": PINS["packet"]["revision"],
"images_verified": len(images),
"published_files_verified": sorted(published),
}
print(json.dumps(report, indent=1))


if __name__ == "__main__":
Expand Down
92 changes: 92 additions & 0 deletions examples/doc-field-extraction-omni/pins.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,92 @@
{
"omni": {
"repo": "getomni-ai/ocr-benchmark",
"revision": "4ed0d95271ca00107726230f7a0944ed9e90d897",
"license": "MIT",
"base_url": "https://huggingface.co/datasets/getomni-ai/ocr-benchmark/resolve/4ed0d95271ca00107726230f7a0944ed9e90d897/test",
"metadata": {
"path": "metadata.jsonl",
"bytes": 8370297,
"sha256": "dad8fc254b20fcef351ccfc36add27fc01e2f426944a30f288769ead82e9814b"
}
},
"packet": {
"repo": "superlinked/sie-task-evidence",
"revision": "f10e813f2206348e31da363da2bc90b06086d7d9",
"folder": "doc-field-extraction-native/2026-10-07",
"base_url": "https://huggingface.co/datasets/superlinked/sie-task-evidence/resolve/f10e813f2206348e31da363da2bc90b06086d7d9/doc-field-extraction-native/2026-10-07",
"manifest_sha256": "69823e6d88f3a5dccd0602427be28b65e0fd9641ac02c17b6cf76256389fbad9",
"files": {
"pilot/grader/A1.jsonl": {
"bytes": 230555,
"sha256": "01bcac420272ffa2097b349e43feafcd7e37e745afb15329515390fc65a6252d"
},
"pilot/grader/omni_scores.json": {
"bytes": 17760,
"sha256": "3cbf515ac15125cfcca56b1d2c4fcb3c3c482b34c4cc73e93ec9f56bfbb97884"
},
"pilot/requests/A1.bodies.show.jsonl": {
"bytes": 745123,
"sha256": "5366928b6b6a910c2d19ea4bfdc4e4d77220a207344a3ece59abe0261c38c040"
},
"pilot/requests/image_manifest.json": {
"bytes": 34028,
"sha256": "99724cbeaef2f9b27330e375c2a7c095b572588a6f15ea445333747e1c9c14ca"
},
"confirm/grader/A1.jsonl": {
"bytes": 1138350,
"sha256": "3cb6dd985c57ab3e67ce707b9fd108fcd35983b5691f52e41742896f5582e809"
},
"confirm/grader/omni_scores.json": {
"bytes": 97585,
"sha256": "c4033d5117cfe606f95d4e8a98cb22a465d51a9b67c288e2d0489e2dcb19f03e"
},
"confirm/requests/A1.confirm.bodies.jsonl.sha256": {
"bytes": 65,
"sha256": "814f56f27140a65fd415313825454716f93b8a3ff4ca9c4cc30d7cec8ebea607"
},
"confirm/requests/A1.confirm.bodies.show.jsonl": {
"bytes": 3941564,
"sha256": "5626ab61415798aac7c1ac3e1f1ff7cd973671ba4be39ba9984119430c35bfe6"
},
"confirm/requests/image_manifest.json": {
"bytes": 153198,
"sha256": "c4fbbc931614139167500544c217798acb2bfb5314d669cd0153eeec28bf5e58"
}
}
},
"sets": {
"pilot": {
"image_manifest": "sets/pilot.image_manifest.json",
"image_manifest_sha256": "99724cbeaef2f9b27330e375c2a7c095b572588a6f15ea445333747e1c9c14ca",
"study_documents": 100,
"study_ids_sha256": "6577f0740b076a4c9c21d07743aed14babaaf5c7dfa01446ea376bafc8059b5d",
"bodies_rows": 106,
"bodies_sha256": "d364ac860a1790772baaa9c343a6bddd6aac055bf06dc45c095d2bd75240396a",
"bodies_sha256_note": "sha256 of the pilot bodies file as sent (100 study rows then 6 unscored display rows); the file itself is not published",
"bodies_show_sha256": "5366928b6b6a910c2d19ea4bfdc4e4d77220a207344a3ece59abe0261c38c040",
"published_mean": 90.15,
"derived_images": {
"omni-407": {
"source_file": "images/407.png",
"source_sha256": "34662c3b94518452d8ce47e41e9949ddc8052575037326d9d36db3875fda5474",
"transform": "Pillow: Image.open(src).convert('RGB').save(out, 'JPEG', quality=90)",
"why": "the source PNG is over 5 MB as base64, so the study re-encoded it once and every arm received the re-encoded bytes",
"reproduced_with": "Pillow 12.2.0"
}
}
},
"confirm": {
"image_manifest": "sets/confirm.image_manifest.json",
"image_manifest_sha256": "c4fbbc931614139167500544c217798acb2bfb5314d669cd0153eeec28bf5e58",
"study_documents": 546,
"study_ids_sha256": "be62dac0164637c6fcae7fc532a63f417b406e11c5f1a60ae4488e8a9d0bef79",
"bodies_rows": 546,
"bodies_sha256": "8e796736eac72b18e0815a6c62e265c29e154768ced92006140aad50802010d8",
"bodies_sha256_note": "published in confirm/requests/A1.confirm.bodies.jsonl.sha256; the bodies file itself is not published",
"bodies_show_sha256": "5626ab61415798aac7c1ac3e1f1ff7cd973671ba4be39ba9984119430c35bfe6",
"published_mean": 91.03,
"derived_images": {}
}
}
}
Loading
Loading