{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:O6BF2CAS2NZMXB3GB2D2NZSK4E","short_pith_number":"pith:O6BF2CAS","schema_version":"1.0","canonical_sha256":"77825d0812d372cb87660e87a6e64ae104ffeb78acf03e5a9219f47d055dc929","source":{"kind":"arxiv","id":"2505.09694","version":2},"attestation_state":"computed","paper":{"title":"EWMBench: Evaluating Scene, Motion, and Semantic Quality in Embodied World Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Guanghui Ren, Hu Yue, Liliang Chen, Maoqing Yao, Pengfei Zhou, Shengcong Chen, Siyuan Huang, Yue Liao","submitted_at":"2025-05-14T18:00:19Z","abstract_excerpt":"Recent advances in creative AI have enabled the synthesis of high-fidelity images and videos conditioned on language instructions. Building on these developments, text-to-video diffusion models have evolved into embodied world models (EWMs) capable of generating physically plausible scenes from language commands, effectively bridging vision and action in embodied AI applications. This work addresses the critical challenge of evaluating EWMs beyond general perceptual metrics to ensure the generation of physically grounded and action-consistent behaviors. We propose the Embodied World Model Benc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.09694","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2025-05-14T18:00:19Z","cross_cats_sorted":[],"title_canon_sha256":"f823565b1e02f13f941e3fd398812e129c0ff8e5fd22211097b75254187493bb","abstract_canon_sha256":"b59e88dd78e07dc0e07c590d7f6926b72cb956d7f2ece4e168d14251cf8314b7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:04:51.612861Z","signature_b64":"7lRRlzt9CW95QTf/koPI8qpxl7nrQ+uFu2gWPDZnkbFG8QBIUyLRqRo5ZYDCo1YJxHwnwW57waFq0/WWcnDpAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"77825d0812d372cb87660e87a6e64ae104ffeb78acf03e5a9219f47d055dc929","last_reissued_at":"2026-07-05T11:04:51.612295Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:04:51.612295Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EWMBench: Evaluating Scene, Motion, and Semantic Quality in Embodied World Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Guanghui Ren, Hu Yue, Liliang Chen, Maoqing Yao, Pengfei Zhou, Shengcong Chen, Siyuan Huang, Yue Liao","submitted_at":"2025-05-14T18:00:19Z","abstract_excerpt":"Recent advances in creative AI have enabled the synthesis of high-fidelity images and videos conditioned on language instructions. Building on these developments, text-to-video diffusion models have evolved into embodied world models (EWMs) capable of generating physically plausible scenes from language commands, effectively bridging vision and action in embodied AI applications. This work addresses the critical challenge of evaluating EWMs beyond general perceptual metrics to ensure the generation of physically grounded and action-consistent behaviors. We propose the Embodied World Model Benc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.09694","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.09694/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.09694","created_at":"2026-07-05T11:04:51.612373+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.09694v2","created_at":"2026-07-05T11:04:51.612373+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.09694","created_at":"2026-07-05T11:04:51.612373+00:00"},{"alias_kind":"pith_short_12","alias_value":"O6BF2CAS2NZM","created_at":"2026-07-05T11:04:51.612373+00:00"},{"alias_kind":"pith_short_16","alias_value":"O6BF2CAS2NZMXB3G","created_at":"2026-07-05T11:04:51.612373+00:00"},{"alias_kind":"pith_short_8","alias_value":"O6BF2CAS","created_at":"2026-07-05T11:04:51.612373+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06401","citing_title":"A Definition and Roadmap for World Models","ref_index":276,"is_internal_anchor":true},{"citing_arxiv_id":"2606.11129","citing_title":"WorldOlympiad: Can Your World Model Survive a Triathlon?","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07687","citing_title":"What Makes Video World Model Latents Action-Relevant: Prediction over Reconstruction","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04463","citing_title":"OSCAR: Omni-Embodiment Action-Conditioned World Model for Robotics","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.15032","citing_title":"How Should World Models Be Evaluated for Embodied Decision-Making? A Decision-Making-Centric Position","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25874","citing_title":"WBench: A Comprehensive Multi-turn Benchmark for Interactive Video World Model Evaluation","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17912","citing_title":"WorldArena 2.0: Extending Embodied World Model Benchmarking on Modality, Functionality and Platform","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2508.05635","citing_title":"Genie Envisioner: A Unified World Foundation Platform for Robotic Manipulation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19092","citing_title":"RoboWM-Bench: A Benchmark for Evaluating World Models in Robotic Manipulation","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12090","citing_title":"World Action Models: The Next Frontier in Embodied AI","ref_index":218,"is_internal_anchor":false},{"citing_arxiv_id":"2511.00062","citing_title":"World Simulation with Video Foundation Models for Physical AI","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06388","citing_title":"Reconstruction or Semantics? What Makes a Latent Space Useful for Robotic World Models","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00080","citing_title":"World Model for Robot Learning: A Comprehensive Survey","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19092","citing_title":"RoboWM-Bench: A Benchmark for Evaluating World Models in Robotic Manipulation","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O6BF2CAS2NZMXB3GB2D2NZSK4E","json":"https://pith.science/pith/O6BF2CAS2NZMXB3GB2D2NZSK4E.json","graph_json":"https://pith.science/api/pith-number/O6BF2CAS2NZMXB3GB2D2NZSK4E/graph.json","events_json":"https://pith.science/api/pith-number/O6BF2CAS2NZMXB3GB2D2NZSK4E/events.json","paper":"https://pith.science/paper/O6BF2CAS"},"agent_actions":{"view_html":"https://pith.science/pith/O6BF2CAS2NZMXB3GB2D2NZSK4E","download_json":"https://pith.science/pith/O6BF2CAS2NZMXB3GB2D2NZSK4E.json","view_paper":"https://pith.science/paper/O6BF2CAS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.09694&json=true","fetch_graph":"https://pith.science/api/pith-number/O6BF2CAS2NZMXB3GB2D2NZSK4E/graph.json","fetch_events":"https://pith.science/api/pith-number/O6BF2CAS2NZMXB3GB2D2NZSK4E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O6BF2CAS2NZMXB3GB2D2NZSK4E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O6BF2CAS2NZMXB3GB2D2NZSK4E/action/storage_attestation","attest_author":"https://pith.science/pith/O6BF2CAS2NZMXB3GB2D2NZSK4E/action/author_attestation","sign_citation":"https://pith.science/pith/O6BF2CAS2NZMXB3GB2D2NZSK4E/action/citation_signature","submit_replication":"https://pith.science/pith/O6BF2CAS2NZMXB3GB2D2NZSK4E/action/replication_record"}},"created_at":"2026-07-05T11:04:51.612373+00:00","updated_at":"2026-07-05T11:04:51.612373+00:00"}