{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EQAS7EHUWQQOTUNVXVSIXIBYBN","short_pith_number":"pith:EQAS7EHU","schema_version":"1.0","canonical_sha256":"24012f90f4b420e9d1b5bd648ba0380b597ec5cb80342dfc84294a87e469d0f8","source":{"kind":"arxiv","id":"2508.01922","version":1},"attestation_state":"computed","paper":{"title":"Beyond Simulation: Benchmarking World Models for Planning and Causality in Autonomous Driving","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Hunter Schofield, Jinjun Shan, Kasra Rezaee, Mohammed Elmahgiubi","submitted_at":"2025-08-03T21:12:21Z","abstract_excerpt":"World models have become increasingly popular in acting as learned traffic simulators. Recent work has explored replacing traditional traffic simulators with world models for policy training. In this work, we explore the robustness of existing metrics to evaluate world models as traffic simulators to see if the same metrics are suitable for evaluating a world model as a pseudo-environment for policy training. Specifically, we analyze the metametric employed by the Waymo Open Sim-Agents Challenge (WOSAC) and compare world model predictions on standard scenarios where the agents are fully or par"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.01922","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2025-08-03T21:12:21Z","cross_cats_sorted":[],"title_canon_sha256":"a68dbf09d45d9812f0abed5a409f70e303f7fb1cd447f51b33555b12ae6d7adf","abstract_canon_sha256":"736970620757fc5d0550879e4e9aa1251af2fdf5bb47158b676df9c8f05b9b75"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:47:48.169885Z","signature_b64":"LV3tDqvyCO6wS93dZtmTXWxWzPHKXO/41ixoLDJe69LjoD4kDSkZomVGwwnzZe4h3bscdSZU7JsS+vjwSQ4bAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"24012f90f4b420e9d1b5bd648ba0380b597ec5cb80342dfc84294a87e469d0f8","last_reissued_at":"2026-07-05T11:47:48.169397Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:47:48.169397Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Simulation: Benchmarking World Models for Planning and Causality in Autonomous Driving","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Hunter Schofield, Jinjun Shan, Kasra Rezaee, Mohammed Elmahgiubi","submitted_at":"2025-08-03T21:12:21Z","abstract_excerpt":"World models have become increasingly popular in acting as learned traffic simulators. Recent work has explored replacing traditional traffic simulators with world models for policy training. In this work, we explore the robustness of existing metrics to evaluate world models as traffic simulators to see if the same metrics are suitable for evaluating a world model as a pseudo-environment for policy training. Specifically, we analyze the metametric employed by the Waymo Open Sim-Agents Challenge (WOSAC) and compare world model predictions on standard scenarios where the agents are fully or par"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.01922","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.01922/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.01922","created_at":"2026-07-05T11:47:48.169461+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.01922v1","created_at":"2026-07-05T11:47:48.169461+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.01922","created_at":"2026-07-05T11:47:48.169461+00:00"},{"alias_kind":"pith_short_12","alias_value":"EQAS7EHUWQQO","created_at":"2026-07-05T11:47:48.169461+00:00"},{"alias_kind":"pith_short_16","alias_value":"EQAS7EHUWQQOTUNV","created_at":"2026-07-05T11:47:48.169461+00:00"},{"alias_kind":"pith_short_8","alias_value":"EQAS7EHU","created_at":"2026-07-05T11:47:48.169461+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23588","citing_title":"A Generative Model for Closed-Loop Microsimulation of Signalized Intersections","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EQAS7EHUWQQOTUNVXVSIXIBYBN","json":"https://pith.science/pith/EQAS7EHUWQQOTUNVXVSIXIBYBN.json","graph_json":"https://pith.science/api/pith-number/EQAS7EHUWQQOTUNVXVSIXIBYBN/graph.json","events_json":"https://pith.science/api/pith-number/EQAS7EHUWQQOTUNVXVSIXIBYBN/events.json","paper":"https://pith.science/paper/EQAS7EHU"},"agent_actions":{"view_html":"https://pith.science/pith/EQAS7EHUWQQOTUNVXVSIXIBYBN","download_json":"https://pith.science/pith/EQAS7EHUWQQOTUNVXVSIXIBYBN.json","view_paper":"https://pith.science/paper/EQAS7EHU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.01922&json=true","fetch_graph":"https://pith.science/api/pith-number/EQAS7EHUWQQOTUNVXVSIXIBYBN/graph.json","fetch_events":"https://pith.science/api/pith-number/EQAS7EHUWQQOTUNVXVSIXIBYBN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EQAS7EHUWQQOTUNVXVSIXIBYBN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EQAS7EHUWQQOTUNVXVSIXIBYBN/action/storage_attestation","attest_author":"https://pith.science/pith/EQAS7EHUWQQOTUNVXVSIXIBYBN/action/author_attestation","sign_citation":"https://pith.science/pith/EQAS7EHUWQQOTUNVXVSIXIBYBN/action/citation_signature","submit_replication":"https://pith.science/pith/EQAS7EHUWQQOTUNVXVSIXIBYBN/action/replication_record"}},"created_at":"2026-07-05T11:47:48.169461+00:00","updated_at":"2026-07-05T11:47:48.169461+00:00"}