{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:A2KBZDEVVQUX6K3ARH73BGFQFS","short_pith_number":"pith:A2KBZDEV","schema_version":"1.0","canonical_sha256":"06941c8c95ac297f2b6089ffb098b02c8107d26c257da6498941e32767d2eedc","source":{"kind":"arxiv","id":"2311.09048","version":3},"attestation_state":"computed","paper":{"title":"GRASP: A novel benchmark for evaluating language GRounding And Situated Physics understanding in multimodal language models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Annika Richter, Cornelius Wolff, Elia Bruni, Mario Holubar, Serwan Jassim, Xenia Ohmer","submitted_at":"2023-11-15T15:38:28Z","abstract_excerpt":"This paper presents GRASP, a novel benchmark to evaluate the language grounding and physical understanding capabilities of video-based multimodal large language models (LLMs). This evaluation is accomplished via a two-tier approach leveraging Unity simulations. The first level tests for language grounding by assessing a model's ability to relate simple textual descriptions with visual information. The second level evaluates the model's understanding of \"Intuitive Physics\" principles, such as object permanence and continuity. In addition to releasing the benchmark, we use it to evaluate several"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.09048","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-11-15T15:38:28Z","cross_cats_sorted":[],"title_canon_sha256":"87a3c033b0bb5e25cd39258f87e37bfeb62b5620e8fc1ed59de6cc16ffb3ca89","abstract_canon_sha256":"9c1264770b9d5b3e60c44f57f3838e7d573383df271dc0536b9a4d08dab29c5d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:28:03.597581Z","signature_b64":"jWbgv+6yu1ZxvtOx0/NBWwT6pxpasYxc/TnIjFCrKzG6+0ZM9O2IlYFheu9LZrYnBIT4q6N5u0Ds7GPXS6W/DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"06941c8c95ac297f2b6089ffb098b02c8107d26c257da6498941e32767d2eedc","last_reissued_at":"2026-07-05T08:28:03.597150Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:28:03.597150Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GRASP: A novel benchmark for evaluating language GRounding And Situated Physics understanding in multimodal language models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Annika Richter, Cornelius Wolff, Elia Bruni, Mario Holubar, Serwan Jassim, Xenia Ohmer","submitted_at":"2023-11-15T15:38:28Z","abstract_excerpt":"This paper presents GRASP, a novel benchmark to evaluate the language grounding and physical understanding capabilities of video-based multimodal large language models (LLMs). This evaluation is accomplished via a two-tier approach leveraging Unity simulations. The first level tests for language grounding by assessing a model's ability to relate simple textual descriptions with visual information. The second level evaluates the model's understanding of \"Intuitive Physics\" principles, such as object permanence and continuity. In addition to releasing the benchmark, we use it to evaluate several"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.09048","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.09048/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.09048","created_at":"2026-07-05T08:28:03.597204+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.09048v3","created_at":"2026-07-05T08:28:03.597204+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.09048","created_at":"2026-07-05T08:28:03.597204+00:00"},{"alias_kind":"pith_short_12","alias_value":"A2KBZDEVVQUX","created_at":"2026-07-05T08:28:03.597204+00:00"},{"alias_kind":"pith_short_16","alias_value":"A2KBZDEVVQUX6K3A","created_at":"2026-07-05T08:28:03.597204+00:00"},{"alias_kind":"pith_short_8","alias_value":"A2KBZDEV","created_at":"2026-07-05T08:28:03.597204+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26781","citing_title":"LiveK12Bench: Have Large Multimodal Models Truly Conquered High School-level Examinations?","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13294","citing_title":"VisPhyWorld: Probing Physical Reasoning via Code-Driven Video Reconstruction","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2501.09038","citing_title":"Do generative video models understand physical principles?","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2410.05363","citing_title":"Towards World Simulator: Crafting Physical Commonsense-Based Benchmark for Video Generation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20328","citing_title":"Video models are zero-shot learners and reasoners","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A2KBZDEVVQUX6K3ARH73BGFQFS","json":"https://pith.science/pith/A2KBZDEVVQUX6K3ARH73BGFQFS.json","graph_json":"https://pith.science/api/pith-number/A2KBZDEVVQUX6K3ARH73BGFQFS/graph.json","events_json":"https://pith.science/api/pith-number/A2KBZDEVVQUX6K3ARH73BGFQFS/events.json","paper":"https://pith.science/paper/A2KBZDEV"},"agent_actions":{"view_html":"https://pith.science/pith/A2KBZDEVVQUX6K3ARH73BGFQFS","download_json":"https://pith.science/pith/A2KBZDEVVQUX6K3ARH73BGFQFS.json","view_paper":"https://pith.science/paper/A2KBZDEV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.09048&json=true","fetch_graph":"https://pith.science/api/pith-number/A2KBZDEVVQUX6K3ARH73BGFQFS/graph.json","fetch_events":"https://pith.science/api/pith-number/A2KBZDEVVQUX6K3ARH73BGFQFS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A2KBZDEVVQUX6K3ARH73BGFQFS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A2KBZDEVVQUX6K3ARH73BGFQFS/action/storage_attestation","attest_author":"https://pith.science/pith/A2KBZDEVVQUX6K3ARH73BGFQFS/action/author_attestation","sign_citation":"https://pith.science/pith/A2KBZDEVVQUX6K3ARH73BGFQFS/action/citation_signature","submit_replication":"https://pith.science/pith/A2KBZDEVVQUX6K3ARH73BGFQFS/action/replication_record"}},"created_at":"2026-07-05T08:28:03.597204+00:00","updated_at":"2026-07-05T08:28:03.597204+00:00"}