{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RF2MKADZKARBXSQYVWV37K3ONS","short_pith_number":"pith:RF2MKADZ","schema_version":"1.0","canonical_sha256":"8974c5007950221bca18adabbfab6e6c9e32b0b3a5bd914833d2e715194df46d","source":{"kind":"arxiv","id":"2405.05956","version":2},"attestation_state":"computed","paper":{"title":"Probing Multimodal LLMs as World Models for Driving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Alaa Maalouf, Daniela Rus, Guy Rosman, Sertac Karaman, Shiva Sreeram, Tsun-Hsuan Wang","submitted_at":"2024-05-09T17:52:42Z","abstract_excerpt":"We provide a sober look at the application of Multimodal Large Language Models (MLLMs) in autonomous driving, challenging common assumptions about their ability to interpret dynamic driving scenarios. Despite advances in models like GPT-4o, their performance in complex driving environments remains largely unexplored. Our experimental study assesses various MLLMs as world models using in-car camera perspectives and reveals that while these models excel at interpreting individual images, they struggle to synthesize coherent narratives across frames, leading to considerable inaccuracies in unders"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.05956","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-05-09T17:52:42Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"b8ad9deb55073d604d09c811a29a5408b4da3edaae25204bc1f74ef3e4f90751","abstract_canon_sha256":"e90bdae8da4993dd281b90c0178211e1959e10dc0a6fd851b2bf632c39b41a36"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:26:26.101045Z","signature_b64":"K3sJDnmKBTNYCAqX8GsQmGcAtdQcy30fCIyJp8lWFaZR79+Aol7dPsgqPUA9rlUBvbdp6seiqacftPDQERVRDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8974c5007950221bca18adabbfab6e6c9e32b0b3a5bd914833d2e715194df46d","last_reissued_at":"2026-07-05T09:26:26.100615Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:26:26.100615Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Probing Multimodal LLMs as World Models for Driving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Alaa Maalouf, Daniela Rus, Guy Rosman, Sertac Karaman, Shiva Sreeram, Tsun-Hsuan Wang","submitted_at":"2024-05-09T17:52:42Z","abstract_excerpt":"We provide a sober look at the application of Multimodal Large Language Models (MLLMs) in autonomous driving, challenging common assumptions about their ability to interpret dynamic driving scenarios. Despite advances in models like GPT-4o, their performance in complex driving environments remains largely unexplored. Our experimental study assesses various MLLMs as world models using in-car camera perspectives and reveals that while these models excel at interpreting individual images, they struggle to synthesize coherent narratives across frames, leading to considerable inaccuracies in unders"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.05956","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.05956/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.05956","created_at":"2026-07-05T09:26:26.100670+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.05956v2","created_at":"2026-07-05T09:26:26.100670+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.05956","created_at":"2026-07-05T09:26:26.100670+00:00"},{"alias_kind":"pith_short_12","alias_value":"RF2MKADZKARB","created_at":"2026-07-05T09:26:26.100670+00:00"},{"alias_kind":"pith_short_16","alias_value":"RF2MKADZKARBXSQY","created_at":"2026-07-05T09:26:26.100670+00:00"},{"alias_kind":"pith_short_8","alias_value":"RF2MKADZ","created_at":"2026-07-05T09:26:26.100670+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RF2MKADZKARBXSQYVWV37K3ONS","json":"https://pith.science/pith/RF2MKADZKARBXSQYVWV37K3ONS.json","graph_json":"https://pith.science/api/pith-number/RF2MKADZKARBXSQYVWV37K3ONS/graph.json","events_json":"https://pith.science/api/pith-number/RF2MKADZKARBXSQYVWV37K3ONS/events.json","paper":"https://pith.science/paper/RF2MKADZ"},"agent_actions":{"view_html":"https://pith.science/pith/RF2MKADZKARBXSQYVWV37K3ONS","download_json":"https://pith.science/pith/RF2MKADZKARBXSQYVWV37K3ONS.json","view_paper":"https://pith.science/paper/RF2MKADZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.05956&json=true","fetch_graph":"https://pith.science/api/pith-number/RF2MKADZKARBXSQYVWV37K3ONS/graph.json","fetch_events":"https://pith.science/api/pith-number/RF2MKADZKARBXSQYVWV37K3ONS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RF2MKADZKARBXSQYVWV37K3ONS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RF2MKADZKARBXSQYVWV37K3ONS/action/storage_attestation","attest_author":"https://pith.science/pith/RF2MKADZKARBXSQYVWV37K3ONS/action/author_attestation","sign_citation":"https://pith.science/pith/RF2MKADZKARBXSQYVWV37K3ONS/action/citation_signature","submit_replication":"https://pith.science/pith/RF2MKADZKARBXSQYVWV37K3ONS/action/replication_record"}},"created_at":"2026-07-05T09:26:26.100670+00:00","updated_at":"2026-07-05T09:26:26.100670+00:00"}