{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KBJQV6W2KQ6I36LKYT4MQ53G6D","short_pith_number":"pith:KBJQV6W2","schema_version":"1.0","canonical_sha256":"50530afada543c8df96ac4f8c87766f0ce8ee1ae674e393d6def41e63117317b","source":{"kind":"arxiv","id":"2312.10728","version":1},"attestation_state":"computed","paper":{"title":"Benchmarks for Physical Reasoning AI","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Andrei Muresanu, Andrew Melnik, Animesh Garg, Helge Ritter, Moritz Lange, Mozhgan Saeidi, Robin Schiewer","submitted_at":"2023-12-17T14:24:03Z","abstract_excerpt":"Physical reasoning is a crucial aspect in the development of general AI systems, given that human learning starts with interacting with the physical world before progressing to more complex concepts. Although researchers have studied and assessed the physical reasoning of AI approaches through various specific benchmarks, there is no comprehensive approach to evaluating and measuring progress. Therefore, we aim to offer an overview of existing benchmarks and their solution approaches and propose a unified perspective for measuring the physical reasoning capacity of AI systems. We select benchm"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.10728","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-12-17T14:24:03Z","cross_cats_sorted":[],"title_canon_sha256":"f4b9de164b83776cef2bb7bb4e43221073f956e216f58fd7638f6f9133e5fa10","abstract_canon_sha256":"04185855331e5c5c080589953f044ec36974b08d744f5d95b294c1282e0c656c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:25:11.394007Z","signature_b64":"zJMMBTZN+1T9QUo4LPKxyz5icn8wK1isusaRCC66Qu0K16qTzLGsrrvn6faNDj3OaUtrYwbW1KvCWVRZdmGPCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"50530afada543c8df96ac4f8c87766f0ce8ee1ae674e393d6def41e63117317b","last_reissued_at":"2026-07-05T07:25:11.393491Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:25:11.393491Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarks for Physical Reasoning AI","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Andrei Muresanu, Andrew Melnik, Animesh Garg, Helge Ritter, Moritz Lange, Mozhgan Saeidi, Robin Schiewer","submitted_at":"2023-12-17T14:24:03Z","abstract_excerpt":"Physical reasoning is a crucial aspect in the development of general AI systems, given that human learning starts with interacting with the physical world before progressing to more complex concepts. Although researchers have studied and assessed the physical reasoning of AI approaches through various specific benchmarks, there is no comprehensive approach to evaluating and measuring progress. Therefore, we aim to offer an overview of existing benchmarks and their solution approaches and propose a unified perspective for measuring the physical reasoning capacity of AI systems. We select benchm"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.10728","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.10728/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.10728","created_at":"2026-07-05T07:25:11.393559+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.10728v1","created_at":"2026-07-05T07:25:11.393559+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.10728","created_at":"2026-07-05T07:25:11.393559+00:00"},{"alias_kind":"pith_short_12","alias_value":"KBJQV6W2KQ6I","created_at":"2026-07-05T07:25:11.393559+00:00"},{"alias_kind":"pith_short_16","alias_value":"KBJQV6W2KQ6I36LK","created_at":"2026-07-05T07:25:11.393559+00:00"},{"alias_kind":"pith_short_8","alias_value":"KBJQV6W2","created_at":"2026-07-05T07:25:11.393559+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07602","citing_title":"Sample-Efficient Post-Training for LEGO Spatial-Physics Reasoning","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13294","citing_title":"VisPhyWorld: Probing Physical Reasoning via Code-Driven Video Reconstruction","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16054","citing_title":"Ada-Diffuser: Latent-Aware Adaptive Diffusion for Decision-Making","ref_index":189,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KBJQV6W2KQ6I36LKYT4MQ53G6D","json":"https://pith.science/pith/KBJQV6W2KQ6I36LKYT4MQ53G6D.json","graph_json":"https://pith.science/api/pith-number/KBJQV6W2KQ6I36LKYT4MQ53G6D/graph.json","events_json":"https://pith.science/api/pith-number/KBJQV6W2KQ6I36LKYT4MQ53G6D/events.json","paper":"https://pith.science/paper/KBJQV6W2"},"agent_actions":{"view_html":"https://pith.science/pith/KBJQV6W2KQ6I36LKYT4MQ53G6D","download_json":"https://pith.science/pith/KBJQV6W2KQ6I36LKYT4MQ53G6D.json","view_paper":"https://pith.science/paper/KBJQV6W2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.10728&json=true","fetch_graph":"https://pith.science/api/pith-number/KBJQV6W2KQ6I36LKYT4MQ53G6D/graph.json","fetch_events":"https://pith.science/api/pith-number/KBJQV6W2KQ6I36LKYT4MQ53G6D/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KBJQV6W2KQ6I36LKYT4MQ53G6D/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KBJQV6W2KQ6I36LKYT4MQ53G6D/action/storage_attestation","attest_author":"https://pith.science/pith/KBJQV6W2KQ6I36LKYT4MQ53G6D/action/author_attestation","sign_citation":"https://pith.science/pith/KBJQV6W2KQ6I36LKYT4MQ53G6D/action/citation_signature","submit_replication":"https://pith.science/pith/KBJQV6W2KQ6I36LKYT4MQ53G6D/action/replication_record"}},"created_at":"2026-07-05T07:25:11.393559+00:00","updated_at":"2026-07-05T07:25:11.393559+00:00"}