{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZLN7HS42MNHZGL3QE6VTE7BUH5","short_pith_number":"pith:ZLN7HS42","schema_version":"1.0","canonical_sha256":"cadbf3cb9a634f932f7027ab327c343f40d2b15bca086d19e14259e9d0dcc45d","source":{"kind":"arxiv","id":"2312.09004","version":1},"attestation_state":"computed","paper":{"title":"Holistic chemical evaluation reveals pitfalls in reaction prediction models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"physics.chem-ph","authors_text":"Andres M. Bran, Jeremy S. Luterbacher, Malte Franke, Philippe Schwaller, Remi Schlama, Victor Sabanza Gil","submitted_at":"2023-12-14T14:54:28Z","abstract_excerpt":"The prediction of chemical reactions has gained significant interest within the machine learning community in recent years, owing to its complexity and crucial applications in chemistry. However, model evaluation for this task has been mostly limited to simple metrics like top-k accuracy, which obfuscates fine details of a model's limitations. Inspired by progress in other fields, we propose a new assessment scheme that builds on top of current approaches, steering towards a more holistic evaluation. We introduce the following key components for this goal: CHORISO, a curated dataset along with"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.09004","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"physics.chem-ph","submitted_at":"2023-12-14T14:54:28Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"81629478a1c7e2283b9563acc4ff0bfc49b89f12827a97824da0c8d7e65ee8a1","abstract_canon_sha256":"8abf78e893adff5ae94df575312fe872e05be616564b0a19e1b56742d4185a68"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:24:13.024250Z","signature_b64":"k/x9eE3dL6proRHOtTMLNcGpkuY2+P++NAhJkpgS0Kj3o3lXkIK4u5MYAx/lcEO4vH5NKbk+jVHLqe6YR8FXAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cadbf3cb9a634f932f7027ab327c343f40d2b15bca086d19e14259e9d0dcc45d","last_reissued_at":"2026-07-05T07:24:13.023782Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:24:13.023782Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Holistic chemical evaluation reveals pitfalls in reaction prediction models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"physics.chem-ph","authors_text":"Andres M. Bran, Jeremy S. Luterbacher, Malte Franke, Philippe Schwaller, Remi Schlama, Victor Sabanza Gil","submitted_at":"2023-12-14T14:54:28Z","abstract_excerpt":"The prediction of chemical reactions has gained significant interest within the machine learning community in recent years, owing to its complexity and crucial applications in chemistry. However, model evaluation for this task has been mostly limited to simple metrics like top-k accuracy, which obfuscates fine details of a model's limitations. Inspired by progress in other fields, we propose a new assessment scheme that builds on top of current approaches, steering towards a more holistic evaluation. We introduce the following key components for this goal: CHORISO, a curated dataset along with"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.09004","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.09004/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.09004","created_at":"2026-07-05T07:24:13.023838+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.09004v1","created_at":"2026-07-05T07:24:13.023838+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.09004","created_at":"2026-07-05T07:24:13.023838+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZLN7HS42MNHZ","created_at":"2026-07-05T07:24:13.023838+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZLN7HS42MNHZGL3Q","created_at":"2026-07-05T07:24:13.023838+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZLN7HS42","created_at":"2026-07-05T07:24:13.023838+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.07022","citing_title":"Self-Driving Datasets: From 20 Million Papers to Nuanced Biomedical Knowledge at Scale","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07022","citing_title":"Self-Driving Datasets: From 20 Million Papers to Nuanced Biomedical Knowledge at Scale","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07022","citing_title":"Self-Driving Datasets: From 20 Million Papers to Nuanced Biomedical Knowledge at Scale","ref_index":61,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZLN7HS42MNHZGL3QE6VTE7BUH5","json":"https://pith.science/pith/ZLN7HS42MNHZGL3QE6VTE7BUH5.json","graph_json":"https://pith.science/api/pith-number/ZLN7HS42MNHZGL3QE6VTE7BUH5/graph.json","events_json":"https://pith.science/api/pith-number/ZLN7HS42MNHZGL3QE6VTE7BUH5/events.json","paper":"https://pith.science/paper/ZLN7HS42"},"agent_actions":{"view_html":"https://pith.science/pith/ZLN7HS42MNHZGL3QE6VTE7BUH5","download_json":"https://pith.science/pith/ZLN7HS42MNHZGL3QE6VTE7BUH5.json","view_paper":"https://pith.science/paper/ZLN7HS42","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.09004&json=true","fetch_graph":"https://pith.science/api/pith-number/ZLN7HS42MNHZGL3QE6VTE7BUH5/graph.json","fetch_events":"https://pith.science/api/pith-number/ZLN7HS42MNHZGL3QE6VTE7BUH5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZLN7HS42MNHZGL3QE6VTE7BUH5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZLN7HS42MNHZGL3QE6VTE7BUH5/action/storage_attestation","attest_author":"https://pith.science/pith/ZLN7HS42MNHZGL3QE6VTE7BUH5/action/author_attestation","sign_citation":"https://pith.science/pith/ZLN7HS42MNHZGL3QE6VTE7BUH5/action/citation_signature","submit_replication":"https://pith.science/pith/ZLN7HS42MNHZGL3QE6VTE7BUH5/action/replication_record"}},"created_at":"2026-07-05T07:24:13.023838+00:00","updated_at":"2026-07-05T07:24:13.023838+00:00"}