{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:W6TYX47PJA5FWM3YTZF4O32WXZ","short_pith_number":"pith:W6TYX47P","schema_version":"1.0","canonical_sha256":"b7a78bf3ef483a5b33789e4bc76f56be57c565d41ec70ed68f923f44e81f39a7","source":{"kind":"arxiv","id":"2312.06553","version":3},"attestation_state":"computed","paper":{"title":"HOI-Diff: Text-Driven Synthesis of 3D Human-Object Interactions using Diffusion Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Deqing Sun, Huaizu Jiang, Varun Jampani, Xiaogang Peng, Yiming Xie, Zizhao Wu","submitted_at":"2023-12-11T17:41:17Z","abstract_excerpt":"We address the problem of generating realistic 3D human-object interactions (HOIs) driven by textual prompts. To this end, we take a modular design and decompose the complex task into simpler sub-tasks. We first develop a dual-branch diffusion model (HOI-DM) to generate both human and object motions conditioned on the input text, and encourage coherent motions by a cross-attention communication module between the human and object motion generation branches. We also develop an affordance prediction diffusion model (APDM) to predict the contacting area between the human and object during the int"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.06553","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-11T17:41:17Z","cross_cats_sorted":[],"title_canon_sha256":"8483f2f69024bfc1006bc21bd512b1cab8058220d3488d989f463971d730743e","abstract_canon_sha256":"e71901e28d8a64e9f4b6b83e266211de7c56ec8b4ab0a8287346e731b2357840"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:32:14.023938Z","signature_b64":"JiOUVw4bLFNtuIqEn8AKcoQt4fzDmZia3ooUAKB8x58uObkVyz8YbwbQlv087h6AbZ9QYWDUgzxzi+Es40+bAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b7a78bf3ef483a5b33789e4bc76f56be57c565d41ec70ed68f923f44e81f39a7","last_reissued_at":"2026-07-05T11:32:14.023439Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:32:14.023439Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HOI-Diff: Text-Driven Synthesis of 3D Human-Object Interactions using Diffusion Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Deqing Sun, Huaizu Jiang, Varun Jampani, Xiaogang Peng, Yiming Xie, Zizhao Wu","submitted_at":"2023-12-11T17:41:17Z","abstract_excerpt":"We address the problem of generating realistic 3D human-object interactions (HOIs) driven by textual prompts. To this end, we take a modular design and decompose the complex task into simpler sub-tasks. We first develop a dual-branch diffusion model (HOI-DM) to generate both human and object motions conditioned on the input text, and encourage coherent motions by a cross-attention communication module between the human and object motion generation branches. We also develop an affordance prediction diffusion model (APDM) to predict the contacting area between the human and object during the int"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.06553","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.06553/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.06553","created_at":"2026-07-05T11:32:14.023494+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.06553v3","created_at":"2026-07-05T11:32:14.023494+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.06553","created_at":"2026-07-05T11:32:14.023494+00:00"},{"alias_kind":"pith_short_12","alias_value":"W6TYX47PJA5F","created_at":"2026-07-05T11:32:14.023494+00:00"},{"alias_kind":"pith_short_16","alias_value":"W6TYX47PJA5FWM3Y","created_at":"2026-07-05T11:32:14.023494+00:00"},{"alias_kind":"pith_short_8","alias_value":"W6TYX47P","created_at":"2026-07-05T11:32:14.023494+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.07880","citing_title":"GIRAF: Towards Generalizable Human Interactions with Articulated Objects","ref_index":44,"is_internal_anchor":true},{"citing_arxiv_id":"2607.08742","citing_title":"ContactMimic: Humanoid Object Interaction via Contact Control","ref_index":38,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22806","citing_title":"Policy-as-Data: Learning Generalizable HOI Diffusion Models from Simulated Physics","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2506.19840","citing_title":"GenHSI: Controllable Generation of Human-Scene Interaction Videos","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13729","citing_title":"Coordinating Multiple Conditions for Trajectory-Controlled Human Motion Generation","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W6TYX47PJA5FWM3YTZF4O32WXZ","json":"https://pith.science/pith/W6TYX47PJA5FWM3YTZF4O32WXZ.json","graph_json":"https://pith.science/api/pith-number/W6TYX47PJA5FWM3YTZF4O32WXZ/graph.json","events_json":"https://pith.science/api/pith-number/W6TYX47PJA5FWM3YTZF4O32WXZ/events.json","paper":"https://pith.science/paper/W6TYX47P"},"agent_actions":{"view_html":"https://pith.science/pith/W6TYX47PJA5FWM3YTZF4O32WXZ","download_json":"https://pith.science/pith/W6TYX47PJA5FWM3YTZF4O32WXZ.json","view_paper":"https://pith.science/paper/W6TYX47P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.06553&json=true","fetch_graph":"https://pith.science/api/pith-number/W6TYX47PJA5FWM3YTZF4O32WXZ/graph.json","fetch_events":"https://pith.science/api/pith-number/W6TYX47PJA5FWM3YTZF4O32WXZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W6TYX47PJA5FWM3YTZF4O32WXZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W6TYX47PJA5FWM3YTZF4O32WXZ/action/storage_attestation","attest_author":"https://pith.science/pith/W6TYX47PJA5FWM3YTZF4O32WXZ/action/author_attestation","sign_citation":"https://pith.science/pith/W6TYX47PJA5FWM3YTZF4O32WXZ/action/citation_signature","submit_replication":"https://pith.science/pith/W6TYX47PJA5FWM3YTZF4O32WXZ/action/replication_record"}},"created_at":"2026-07-05T11:32:14.023494+00:00","updated_at":"2026-07-05T11:32:14.023494+00:00"}