{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3PTFDW3A332I77XQA4UOFS3V3G","short_pith_number":"pith:3PTFDW3A","schema_version":"1.0","canonical_sha256":"dbe651db60def48ffef00728e2cb75d984d43451d2d8f5a01cb4651f3ccc5669","source":{"kind":"arxiv","id":"2409.03643","version":2},"attestation_state":"computed","paper":{"title":"Image Over Text: Transforming Formula Recognition Evaluation with Character Detection Matching","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Bin Wang, Bo Zhang, Conghui He, Fan Wu, Linke Ouyang, Renqiu Xia, Rui Zhang, Zhuangcheng Gu","submitted_at":"2024-09-05T16:01:21Z","abstract_excerpt":"Formula recognition presents significant challenges due to the complicated structure and varied notation of mathematical expressions. Despite continuous advancements in formula recognition models, the evaluation metrics employed by these models, such as BLEU and Edit Distance, still exhibit notable limitations. They overlook the fact that the same formula has diverse representations and is highly sensitive to the distribution of training data, thereby causing unfairness in formula recognition evaluation. To this end, we propose a Character Detection Matching (CDM) metric, ensuring the evaluati"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.03643","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-09-05T16:01:21Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"da2626ae7482b1f3f97d771bb57d5b239c9bc75ec5f7f8daf84246f6b5d8054a","abstract_canon_sha256":"9d78760deee935aeacadb5cb86bb727a4dcfbc18b7600cde5a605eb7071ab581"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:38:21.296611Z","signature_b64":"yUPoG/X3FzFe/mhGU9jlfXYAxlPdFwu07Tt4IAD5EEgvggw9qQpB8Eh6TY6xWOSvCFA1shjEP6balolrlLwvAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dbe651db60def48ffef00728e2cb75d984d43451d2d8f5a01cb4651f3ccc5669","last_reissued_at":"2026-07-05T10:38:21.296161Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:38:21.296161Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Image Over Text: Transforming Formula Recognition Evaluation with Character Detection Matching","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Bin Wang, Bo Zhang, Conghui He, Fan Wu, Linke Ouyang, Renqiu Xia, Rui Zhang, Zhuangcheng Gu","submitted_at":"2024-09-05T16:01:21Z","abstract_excerpt":"Formula recognition presents significant challenges due to the complicated structure and varied notation of mathematical expressions. Despite continuous advancements in formula recognition models, the evaluation metrics employed by these models, such as BLEU and Edit Distance, still exhibit notable limitations. They overlook the fact that the same formula has diverse representations and is highly sensitive to the distribution of training data, thereby causing unfairness in formula recognition evaluation. To this end, we propose a Character Detection Matching (CDM) metric, ensuring the evaluati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.03643","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.03643/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.03643","created_at":"2026-07-05T10:38:21.296217+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.03643v2","created_at":"2026-07-05T10:38:21.296217+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.03643","created_at":"2026-07-05T10:38:21.296217+00:00"},{"alias_kind":"pith_short_12","alias_value":"3PTFDW3A332I","created_at":"2026-07-05T10:38:21.296217+00:00"},{"alias_kind":"pith_short_16","alias_value":"3PTFDW3A332I77XQ","created_at":"2026-07-05T10:38:21.296217+00:00"},{"alias_kind":"pith_short_8","alias_value":"3PTFDW3A","created_at":"2026-07-05T10:38:21.296217+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24447","citing_title":"P-MTP: Efficient Document Parsing via Multi-Token Prediction with Progressive Depth Scaling","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22100","citing_title":"MPDocBench-Parse: Benchmarking Practical Multi-page Document Parsing","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2410.21169","citing_title":"Document Parsing Unveiled: Techniques, Challenges, and Prospects for Structured Information Extraction","ref_index":239,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22100","citing_title":"MPDocBench-Parse: Benchmarking Practical Multi-page Document Parsing","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18839","citing_title":"MinerU: An Open-Source Solution for Precise Document Content Extraction","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07492","citing_title":"How Far Is Document Parsing from Solved? PureDocBench: A Source-TraceableBenchmark across Clean, Degraded, and Real-World Settings","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3PTFDW3A332I77XQA4UOFS3V3G","json":"https://pith.science/pith/3PTFDW3A332I77XQA4UOFS3V3G.json","graph_json":"https://pith.science/api/pith-number/3PTFDW3A332I77XQA4UOFS3V3G/graph.json","events_json":"https://pith.science/api/pith-number/3PTFDW3A332I77XQA4UOFS3V3G/events.json","paper":"https://pith.science/paper/3PTFDW3A"},"agent_actions":{"view_html":"https://pith.science/pith/3PTFDW3A332I77XQA4UOFS3V3G","download_json":"https://pith.science/pith/3PTFDW3A332I77XQA4UOFS3V3G.json","view_paper":"https://pith.science/paper/3PTFDW3A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.03643&json=true","fetch_graph":"https://pith.science/api/pith-number/3PTFDW3A332I77XQA4UOFS3V3G/graph.json","fetch_events":"https://pith.science/api/pith-number/3PTFDW3A332I77XQA4UOFS3V3G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3PTFDW3A332I77XQA4UOFS3V3G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3PTFDW3A332I77XQA4UOFS3V3G/action/storage_attestation","attest_author":"https://pith.science/pith/3PTFDW3A332I77XQA4UOFS3V3G/action/author_attestation","sign_citation":"https://pith.science/pith/3PTFDW3A332I77XQA4UOFS3V3G/action/citation_signature","submit_replication":"https://pith.science/pith/3PTFDW3A332I77XQA4UOFS3V3G/action/replication_record"}},"created_at":"2026-07-05T10:38:21.296217+00:00","updated_at":"2026-07-05T10:38:21.296217+00:00"}