{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ECA3QGXSRUH4XGQB2ZH6KGMAGT","short_pith_number":"pith:ECA3QGXS","schema_version":"1.0","canonical_sha256":"2081b81af28d0fcb9a01d64fe5198034c7e64e98539a11c8c37eb56a6ffcc8d6","source":{"kind":"arxiv","id":"2504.15932","version":1},"attestation_state":"computed","paper":{"title":"Reasoning Physical Video Generation with Diffusion Timestep Tokens via Reinforcement Learning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fei Wu, Hanwang Zhang, Jingyuan Chen, Kaihang Pan, Liyu Jia, Wang Lin, Wei Zhao, Wentao Hu, Zhongqi Yue","submitted_at":"2025-04-22T14:20:59Z","abstract_excerpt":"Despite recent progress in video generation, producing videos that adhere to physical laws remains a significant challenge. Traditional diffusion-based methods struggle to extrapolate to unseen physical conditions (eg, velocity) due to their reliance on data-driven approximations. To address this, we propose to integrate symbolic reasoning and reinforcement learning to enforce physical consistency in video generation. We first introduce the Diffusion Timestep Tokenizer (DDT), which learns discrete, recursive visual tokens by recovering visual attributes lost during the diffusion process. The r"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.15932","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-04-22T14:20:59Z","cross_cats_sorted":[],"title_canon_sha256":"a8c57ac500bc8c1172bbc3a568fe8b11db3e1b21126d88aeffdd98505b418e71","abstract_canon_sha256":"4b625af1c740af9ed2376c0a25ca6b03b582b3dc785a21eb9e5543b314c6fe48"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:52:32.161056Z","signature_b64":"Uo0i1qgQL/ChpKzJLEZw9Quh1S04Z0a999IksL8woHwPHadYrPdILFNSlOZTu+AOew5PlB7JD6NRIzS9Kog9Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2081b81af28d0fcb9a01d64fe5198034c7e64e98539a11c8c37eb56a6ffcc8d6","last_reissued_at":"2026-07-05T10:52:32.160589Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:52:32.160589Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reasoning Physical Video Generation with Diffusion Timestep Tokens via Reinforcement Learning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fei Wu, Hanwang Zhang, Jingyuan Chen, Kaihang Pan, Liyu Jia, Wang Lin, Wei Zhao, Wentao Hu, Zhongqi Yue","submitted_at":"2025-04-22T14:20:59Z","abstract_excerpt":"Despite recent progress in video generation, producing videos that adhere to physical laws remains a significant challenge. Traditional diffusion-based methods struggle to extrapolate to unseen physical conditions (eg, velocity) due to their reliance on data-driven approximations. To address this, we propose to integrate symbolic reasoning and reinforcement learning to enforce physical consistency in video generation. We first introduce the Diffusion Timestep Tokenizer (DDT), which learns discrete, recursive visual tokens by recovering visual attributes lost during the diffusion process. The r"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.15932","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.15932/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.15932","created_at":"2026-07-05T10:52:32.160646+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.15932v1","created_at":"2026-07-05T10:52:32.160646+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.15932","created_at":"2026-07-05T10:52:32.160646+00:00"},{"alias_kind":"pith_short_12","alias_value":"ECA3QGXSRUH4","created_at":"2026-07-05T10:52:32.160646+00:00"},{"alias_kind":"pith_short_16","alias_value":"ECA3QGXSRUH4XGQB","created_at":"2026-07-05T10:52:32.160646+00:00"},{"alias_kind":"pith_short_8","alias_value":"ECA3QGXS","created_at":"2026-07-05T10:52:32.160646+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25306","citing_title":"Physics Question Scene Graph: Fine-grained Evaluation of Physical Plausibility in Text-to-Video Generation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18601","citing_title":"Incantation: Natural Language as the Action Interface for Multi-Entity Video World Models","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2509.24702","citing_title":"Enhancing Physical Plausibility in Video Generation by Reasoning the Implausibility","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2603.22844","citing_title":"PhySe-RPO: Physics and Semantics Guided Relative Policy Optimization for Diffusion-Based Surgical Smoke Removal","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ECA3QGXSRUH4XGQB2ZH6KGMAGT","json":"https://pith.science/pith/ECA3QGXSRUH4XGQB2ZH6KGMAGT.json","graph_json":"https://pith.science/api/pith-number/ECA3QGXSRUH4XGQB2ZH6KGMAGT/graph.json","events_json":"https://pith.science/api/pith-number/ECA3QGXSRUH4XGQB2ZH6KGMAGT/events.json","paper":"https://pith.science/paper/ECA3QGXS"},"agent_actions":{"view_html":"https://pith.science/pith/ECA3QGXSRUH4XGQB2ZH6KGMAGT","download_json":"https://pith.science/pith/ECA3QGXSRUH4XGQB2ZH6KGMAGT.json","view_paper":"https://pith.science/paper/ECA3QGXS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.15932&json=true","fetch_graph":"https://pith.science/api/pith-number/ECA3QGXSRUH4XGQB2ZH6KGMAGT/graph.json","fetch_events":"https://pith.science/api/pith-number/ECA3QGXSRUH4XGQB2ZH6KGMAGT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ECA3QGXSRUH4XGQB2ZH6KGMAGT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ECA3QGXSRUH4XGQB2ZH6KGMAGT/action/storage_attestation","attest_author":"https://pith.science/pith/ECA3QGXSRUH4XGQB2ZH6KGMAGT/action/author_attestation","sign_citation":"https://pith.science/pith/ECA3QGXSRUH4XGQB2ZH6KGMAGT/action/citation_signature","submit_replication":"https://pith.science/pith/ECA3QGXSRUH4XGQB2ZH6KGMAGT/action/replication_record"}},"created_at":"2026-07-05T10:52:32.160646+00:00","updated_at":"2026-07-05T10:52:32.160646+00:00"}