""" omr_ned.py — OMR-NED for the Sheetly benchmark bundle, via the reference implementation rather than our own code. OMR-NED (OMR Normalized Edit Distance) is the metric introduced in "Sheet Music Benchmark: Standardized Optical Music Recognition Evaluation" (arXiv:2506.10488, ISMIR 2025). It scores the WHOLE notation — note heads, beams, dots, accidentals, ties, slurs, articulations, ornaments, clefs, key and time signatures, dynamics, directions, lyrics — as a normalized edit distance: (insertions + deletions) / (symbols in prediction + ground truth). 0.0 is a perfect transcription; lower is better. This script contains NO scoring logic of its own. Every comparison is done by musicdiff (github.com/gregchapman-dev/musicdiff), the diff engine the paper's authors used, run as `python -m musicdiff -o omrned` at its default settings. We only cut the ground truth to page one (the same page-one policy score.py uses, measure counts from manifest.json), loop over the pieces, and aggregate. Setup: pip install musicdiff (pulls music21; any Python >= 3.10) Run: python3 omr_ned.py . (from the bundle root; ~15 min, or set OMR_NED_WORKERS to parallelize) A missing or unparseable transcription counts as 1.0 — everything deleted — and the aggregate line reports how many pieces were actually scored. The corpus figure is sum(edit distance) / sum(symbols) over scored pieces; the per-piece numbers are written to omrned-results.json. """ import json, os, statistics, subprocess, sys import xml.etree.ElementTree as ET from concurrent.futures import ThreadPoolExecutor, as_completed ROOT = os.path.abspath(sys.argv[1] if len(sys.argv) > 1 else ".") WORKERS = int(os.environ.get("OMR_NED_WORKERS", "4")) ENGINES = ("sheetly", "audiveris", "oemer") SUITES = ("clean31", "photo31") def cut_ground_truth(manifest): """gt-p1//.musicxml: the ground truth truncated to the page-one measure count, since page one is all any engine was given.""" for suite in SUITES: os.makedirs(f"{ROOT}/gt-p1/{suite}", exist_ok=True) for piece in manifest["suites"][suite]["pieces"]: slug, n = piece["slug"], piece["pages"][0]["measures"] out = f"{ROOT}/gt-p1/{suite}/{slug}.musicxml" if os.path.exists(out): continue tree = ET.parse(f"{ROOT}/{suite}/ground-truth/{slug}.musicxml") for part in tree.getroot().iter(): if part.tag.endswith("part") and not part.tag.endswith("score-part"): for extra in [e for e in list(part) if e.tag.endswith("measure")][n:]: part.remove(extra) tree.write(out, encoding="unicode", xml_declaration=True) def score_one(job): suite, eng, slug = job pred = f"{ROOT}/{suite}/{eng}/{slug}/p1.musicxml" gt = f"{ROOT}/gt-p1/{suite}/{slug}.musicxml" if not os.path.exists(pred): return suite, eng, slug, {"status": "missing", "OMR-NED": 1.0} try: r = subprocess.run([sys.executable, "-m", "musicdiff", gt, pred, "-o", "omrned"], capture_output=True, text=True, timeout=1800) blob = r.stdout[r.stdout.find("{"): r.stdout.rfind("}") + 1] d = json.loads(blob) return suite, eng, slug, { "status": "ok", "OMR-ED": int(d["OMR-ED"]), "pred_symbols": int(d["numSymbolsInPredicted"]), "gt_symbols": int(d["numSymbolsInGroundTruth"]), "OMR-NED": float(d["OMR-NED"]), } except Exception as exc: # engine output musicdiff/music21 cannot parse return suite, eng, slug, {"status": "error", "error": str(exc)[:200], "OMR-NED": 1.0} def main(): manifest = json.load(open(f"{ROOT}/manifest.json")) cut_ground_truth(manifest) jobs = [(suite, eng, piece["slug"]) for suite in SUITES for piece in manifest["suites"][suite]["pieces"] for eng in ENGINES] results = {} with ThreadPoolExecutor(max_workers=WORKERS) as pool: futures = [pool.submit(score_one, j) for j in jobs] for i, fut in enumerate(as_completed(futures), 1): suite, eng, slug, d = fut.result() results.setdefault(suite, {}).setdefault(eng, {})[slug] = d print(f"{i}/{len(jobs)} {suite:8s} {eng:10s} {slug:28s} {d['OMR-NED']:.4f}", flush=True) json.dump(results, open(f"{ROOT}/omrned-results.json", "w"), indent=1) print("\n== OMR-NED (lower is better; 0.0 = perfect) ==") for suite in SUITES: for eng in ENGINES: rows = results[suite][eng].values() ok = [d for d in rows if d["status"] == "ok"] ed = sum(d["OMR-ED"] for d in ok) sym = sum(d["pred_symbols"] + d["gt_symbols"] for d in ok) corpus = ed / sym if sym else 1.0 mean_all = statistics.mean(d["OMR-NED"] for d in rows) med_ok = statistics.median(d["OMR-NED"] for d in ok) if ok else 1.0 print(f"{suite:8s} {eng:10s} corpus {corpus:.4f} " f"mean(all, missing=1.0) {mean_all:.4f} " f"median(scored) {med_ok:.4f} scored {len(ok)}/{len(rows)}") if __name__ == "__main__": main()