{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6LJX6CDU6HV662OA6AVKIKXS2Q","short_pith_number":"pith:6LJX6CDU","schema_version":"1.0","canonical_sha256":"f2d37f0874f1ebef69c0f02aa42af2d43e87ee828b9c4863bb7c82d98715a0ea","source":{"kind":"arxiv","id":"2410.18072","version":1},"attestation_state":"computed","paper":{"title":"WorldSimBench: Towards Video Generation Models as World Simulators","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Enshen Zhou, Jing Shao, Jiwen Yu, Lei Bai, Lijun Li, Lu Sheng, Ruimao Zhang, Wanli Ouyang, Xihui Liu, Xijun Wang, Yiran Qin, Zhelun Shi, Zhenfei Yin","submitted_at":"2024-10-23T17:56:11Z","abstract_excerpt":"Recent advancements in predictive models have demonstrated exceptional capabilities in predicting the future state of objects and scenes. However, the lack of categorization based on inherent characteristics continues to hinder the progress of predictive model development. Additionally, existing benchmarks are unable to effectively evaluate higher-capability, highly embodied predictive models from an embodied perspective. In this work, we classify the functionalities of predictive models into a hierarchy and take the first step in evaluating World Simulators by proposing a dual evaluation fram"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.18072","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-23T17:56:11Z","cross_cats_sorted":[],"title_canon_sha256":"16fae7efcf1212970699aafcbf73582efc0d43f158f647c5d6057ca7784e6ff5","abstract_canon_sha256":"3a8d791d9286e57e7361567254eb3cdfebb42fbca50736cacbf79afbc10b7fcd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:24:48.860216Z","signature_b64":"uDHVyTynuijx0zpk5UZVT/vhMaenMjdl36dczbfi6FWS9dv2brkr8fIJGl0VmMij1QVqMap4VqO0Va01xMDBAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f2d37f0874f1ebef69c0f02aa42af2d43e87ee828b9c4863bb7c82d98715a0ea","last_reissued_at":"2026-07-05T09:24:48.859762Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:24:48.859762Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WorldSimBench: Towards Video Generation Models as World Simulators","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Enshen Zhou, Jing Shao, Jiwen Yu, Lei Bai, Lijun Li, Lu Sheng, Ruimao Zhang, Wanli Ouyang, Xihui Liu, Xijun Wang, Yiran Qin, Zhelun Shi, Zhenfei Yin","submitted_at":"2024-10-23T17:56:11Z","abstract_excerpt":"Recent advancements in predictive models have demonstrated exceptional capabilities in predicting the future state of objects and scenes. However, the lack of categorization based on inherent characteristics continues to hinder the progress of predictive model development. Additionally, existing benchmarks are unable to effectively evaluate higher-capability, highly embodied predictive models from an embodied perspective. In this work, we classify the functionalities of predictive models into a hierarchy and take the first step in evaluating World Simulators by proposing a dual evaluation fram"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.18072","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.18072/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.18072","created_at":"2026-07-05T09:24:48.859817+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.18072v1","created_at":"2026-07-05T09:24:48.859817+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.18072","created_at":"2026-07-05T09:24:48.859817+00:00"},{"alias_kind":"pith_short_12","alias_value":"6LJX6CDU6HV6","created_at":"2026-07-05T09:24:48.859817+00:00"},{"alias_kind":"pith_short_16","alias_value":"6LJX6CDU6HV662OA","created_at":"2026-07-05T09:24:48.859817+00:00"},{"alias_kind":"pith_short_8","alias_value":"6LJX6CDU","created_at":"2026-07-05T09:24:48.859817+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":25,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06401","citing_title":"A Definition and Roadmap for World Models","ref_index":275,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20781","citing_title":"World Action Models: A Survey","ref_index":136,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17566","citing_title":"AoiZora: Topology-Aware Auto-Parallel Optimization for Inference of Diffusion Transformers","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27537","citing_title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17730","citing_title":"ActWorld: From Explorable to Interactive World Model via Action-Aware Memory","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11129","citing_title":"WorldOlympiad: Can Your World Model Survive a Triathlon?","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01896","citing_title":"Divide and Conquer: Decoupled Representation Alignment for Multimodal World Models","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27537","citing_title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04811","citing_title":"Dream.exe: Can Video Generation Models Dream Executable Robot Manipulation?","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27537","citing_title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27537","citing_title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24578","citing_title":"World Models as Group Actions","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28385","citing_title":"RoboGaze: Evaluating Robot World Models via Structured Vision-Language Analysis","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.15032","citing_title":"How Should World Models Be Evaluated for Embodied Decision-Making? A Decision-Making-Centric Position","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25874","citing_title":"WBench: A Comprehensive Multi-turn Benchmark for Interactive Video World Model Evaluation","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27589","citing_title":"What-If World: A Causal Benchmark for General World Models in Embodied Scenarios","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30346","citing_title":"YoCausal: How Far is Video Generation from World Model? A Causality Perspective","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17912","citing_title":"WorldArena 2.0: Extending Embodied World Model Benchmarking on Modality, Functionality and Platform","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2511.18373","citing_title":"MASS: Motion-Aware Spatial-Temporal Grounding for Physics Reasoning and Comprehension in Vision-Language Models","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2512.04678","citing_title":"Reward Forcing: Efficient Streaming Video Generation with Rewarded Distribution Matching Distillation","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12090","citing_title":"World Action Models: The Next Frontier in Embodied AI","ref_index":219,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10434","citing_title":"WorldReasonBench: Human-Aligned Stress Testing of Video Generators as Future World-State Predictors","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01950","citing_title":"TRAP: Tail-aware Ranking Attack for World-Model Planning","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11789","citing_title":"LMMs Meet Object-Centric Vision: Understanding, Segmentation, Editing and Generation","ref_index":131,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01896","citing_title":"Divide and Conquer: Decoupled Representation Alignment for Multimodal World Models","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6LJX6CDU6HV662OA6AVKIKXS2Q","json":"https://pith.science/pith/6LJX6CDU6HV662OA6AVKIKXS2Q.json","graph_json":"https://pith.science/api/pith-number/6LJX6CDU6HV662OA6AVKIKXS2Q/graph.json","events_json":"https://pith.science/api/pith-number/6LJX6CDU6HV662OA6AVKIKXS2Q/events.json","paper":"https://pith.science/paper/6LJX6CDU"},"agent_actions":{"view_html":"https://pith.science/pith/6LJX6CDU6HV662OA6AVKIKXS2Q","download_json":"https://pith.science/pith/6LJX6CDU6HV662OA6AVKIKXS2Q.json","view_paper":"https://pith.science/paper/6LJX6CDU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.18072&json=true","fetch_graph":"https://pith.science/api/pith-number/6LJX6CDU6HV662OA6AVKIKXS2Q/graph.json","fetch_events":"https://pith.science/api/pith-number/6LJX6CDU6HV662OA6AVKIKXS2Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6LJX6CDU6HV662OA6AVKIKXS2Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6LJX6CDU6HV662OA6AVKIKXS2Q/action/storage_attestation","attest_author":"https://pith.science/pith/6LJX6CDU6HV662OA6AVKIKXS2Q/action/author_attestation","sign_citation":"https://pith.science/pith/6LJX6CDU6HV662OA6AVKIKXS2Q/action/citation_signature","submit_replication":"https://pith.science/pith/6LJX6CDU6HV662OA6AVKIKXS2Q/action/replication_record"}},"created_at":"2026-07-05T09:24:48.859817+00:00","updated_at":"2026-07-05T09:24:48.859817+00:00"}