{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IV6HMHGITOVTAGINY7C4JIHVGP","short_pith_number":"pith:IV6HMHGI","schema_version":"1.0","canonical_sha256":"457c761cc89bab30190dc7c5c4a0f533fa9a06c80605360c20fda57011cb884d","source":{"kind":"arxiv","id":"2503.03689","version":1},"attestation_state":"computed","paper":{"title":"DualDiff+: Dual-Branch Diffusion for High-Fidelity Video Generation with Reward Guidance","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Gongpeng Zhao, Lingsi Zhu, Longjun Liu, Ruohong Yu, Weixiang Xu, Xiaofan Li, Zezhong Qian, Zhao Yang","submitted_at":"2025-03-05T17:31:45Z","abstract_excerpt":"Accurate and high-fidelity driving scene reconstruction demands the effective utilization of comprehensive scene information as conditional inputs. Existing methods predominantly rely on 3D bounding boxes and BEV road maps for foreground and background control, which fail to capture the full complexity of driving scenes and adequately integrate multimodal information. In this work, we present DualDiff, a dual-branch conditional diffusion model designed to enhance driving scene generation across multiple views and video sequences. Specifically, we introduce Occupancy Ray-shape Sampling (ORS) as"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.03689","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-03-05T17:31:45Z","cross_cats_sorted":[],"title_canon_sha256":"db06a4eb6c33abec9ad84313cfbddf6100787afffd50670dd07b1bff4c51fc03","abstract_canon_sha256":"b426018dca218393755544ae02287181d4a573df3a903e6478f8093c33f3f43c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:24:58.919692Z","signature_b64":"aw8AoJdGlY7mJjH+5S7zaerPlfXlVsK3tLe74XoIEfy5PvfF/po+jqKnTZMuDqSwaIKpsDLQ+LT2u3vmEPKIDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"457c761cc89bab30190dc7c5c4a0f533fa9a06c80605360c20fda57011cb884d","last_reissued_at":"2026-07-05T10:24:58.919046Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:24:58.919046Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DualDiff+: Dual-Branch Diffusion for High-Fidelity Video Generation with Reward Guidance","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Gongpeng Zhao, Lingsi Zhu, Longjun Liu, Ruohong Yu, Weixiang Xu, Xiaofan Li, Zezhong Qian, Zhao Yang","submitted_at":"2025-03-05T17:31:45Z","abstract_excerpt":"Accurate and high-fidelity driving scene reconstruction demands the effective utilization of comprehensive scene information as conditional inputs. Existing methods predominantly rely on 3D bounding boxes and BEV road maps for foreground and background control, which fail to capture the full complexity of driving scenes and adequately integrate multimodal information. In this work, we present DualDiff, a dual-branch conditional diffusion model designed to enhance driving scene generation across multiple views and video sequences. Specifically, we introduce Occupancy Ray-shape Sampling (ORS) as"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.03689","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.03689/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.03689","created_at":"2026-07-05T10:24:58.919117+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.03689v1","created_at":"2026-07-05T10:24:58.919117+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.03689","created_at":"2026-07-05T10:24:58.919117+00:00"},{"alias_kind":"pith_short_12","alias_value":"IV6HMHGITOVT","created_at":"2026-07-05T10:24:58.919117+00:00"},{"alias_kind":"pith_short_16","alias_value":"IV6HMHGITOVTAGIN","created_at":"2026-07-05T10:24:58.919117+00:00"},{"alias_kind":"pith_short_8","alias_value":"IV6HMHGI","created_at":"2026-07-05T10:24:58.919117+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.29935","citing_title":"CityGen: Structure-Guided City-Style Synthesis for Cross-City Autonomous Driving","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2504.18576","citing_title":"DriVerse: Navigation World Model for Driving Simulation via Multimodal Trajectory Prompting and Motion Alignment","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08546","citing_title":"When Numbers Speak: Aligning Textual Numerals and Visual Instances in Text-to-Video Diffusion Models","ref_index":69,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IV6HMHGITOVTAGINY7C4JIHVGP","json":"https://pith.science/pith/IV6HMHGITOVTAGINY7C4JIHVGP.json","graph_json":"https://pith.science/api/pith-number/IV6HMHGITOVTAGINY7C4JIHVGP/graph.json","events_json":"https://pith.science/api/pith-number/IV6HMHGITOVTAGINY7C4JIHVGP/events.json","paper":"https://pith.science/paper/IV6HMHGI"},"agent_actions":{"view_html":"https://pith.science/pith/IV6HMHGITOVTAGINY7C4JIHVGP","download_json":"https://pith.science/pith/IV6HMHGITOVTAGINY7C4JIHVGP.json","view_paper":"https://pith.science/paper/IV6HMHGI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.03689&json=true","fetch_graph":"https://pith.science/api/pith-number/IV6HMHGITOVTAGINY7C4JIHVGP/graph.json","fetch_events":"https://pith.science/api/pith-number/IV6HMHGITOVTAGINY7C4JIHVGP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IV6HMHGITOVTAGINY7C4JIHVGP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IV6HMHGITOVTAGINY7C4JIHVGP/action/storage_attestation","attest_author":"https://pith.science/pith/IV6HMHGITOVTAGINY7C4JIHVGP/action/author_attestation","sign_citation":"https://pith.science/pith/IV6HMHGITOVTAGINY7C4JIHVGP/action/citation_signature","submit_replication":"https://pith.science/pith/IV6HMHGITOVTAGINY7C4JIHVGP/action/replication_record"}},"created_at":"2026-07-05T10:24:58.919117+00:00","updated_at":"2026-07-05T10:24:58.919117+00:00"}