{"as_of":"2026-08-07T18:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1fd68cd23445af6583bd34e6f14f02ab8aa35ad53107a5c5f6c9cc4154f00e38","coverage":[{"denominator":55,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":55,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T23:58:48.341938Z","state":"measured"},{"denominator":59,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":59,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-29T04:18:02.341742Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T09:39:46.910380Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"cited_work":{"arxiv_id":"2506.15483","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.15483","snapshot_observed_at":"2026-07-04T09:39:46.910380Z","title":"Task-oriented human- object interactions generation with implicit neural representations","venue":null,"work_id":"886fb306-daad-46e3-9245-7a3020fd7314","year":2025},"citing_paper":{"arxiv_id":"2604.04843","last_updated":"2026-04-06T16:44:02Z","snapshot_observed_at":"2026-08-02T08:09:49.884908Z","submitted_at":"2026-04-06T16:44:02Z","title":"InfBaGel: Human-Object-Scene Interaction Generation with Dynamic Perception and Iterative Refinement","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T20:10:16.038861Z"},"links":{"cited_paper":"/paper/2506.15483","citing_paper":"/paper/2604.04843"},"observation_digest":"sha256:ee53af46160c48a0420ea912d5ce07238287b215675fb7f4ac41e393e50586ff","observation_id":"93607517-7bf4-458b-976a-33425fd1a96f","resolution":{"observed_at":"2026-05-10T22:10:50.053682Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"cited_work":{"arxiv_id":"2506.15483","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.15483","snapshot_observed_at":"2026-07-04T09:39:46.910380Z","title":"Task-oriented human- object interactions generation with implicit neural representations","venue":null,"work_id":"886fb306-daad-46e3-9245-7a3020fd7314","year":2025},"citing_paper":{"arxiv_id":"2604.27491","last_updated":"2026-04-30T06:44:10Z","snapshot_observed_at":"2026-07-06T23:12:57.730833Z","submitted_at":"2026-04-30T06:44:10Z","title":"Uni-HOI:A Unified framework for Learning the Joint distribution of Text and Human-Object Interaction","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-07T09:40:37.124547Z"},"links":{"cited_paper":"/paper/2506.15483","citing_paper":"/paper/2604.27491"},"observation_digest":"sha256:5671a3ceff7015a6d9296201ebe02c0a68fecb67b50b0364d6fc9bb0c07556f1","observation_id":"054e805a-e905-4a7f-b2fa-63ea0548617c","resolution":{"observed_at":"2026-05-12T09:41:27.149685Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"cited_work":{"arxiv_id":"2506.15483","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.15483","snapshot_observed_at":"2026-07-04T09:39:46.910380Z","title":"Task-oriented human- object interactions generation with implicit neural representations","venue":null,"work_id":"886fb306-daad-46e3-9245-7a3020fd7314","year":2025},"citing_paper":{"arxiv_id":"2606.22806","last_updated":"2026-06-22T03:32:35Z","snapshot_observed_at":"2026-07-06T23:57:38.036151Z","submitted_at":"2026-06-22T03:32:35Z","title":"Policy-as-Data: Learning Generalizable HOI Diffusion Models from Simulated Physics","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-26T09:40:37.183422Z"},"links":{"cited_paper":"/paper/2506.15483","citing_paper":"/paper/2606.22806"},"observation_digest":"sha256:f06f5edf6f730f6356e86a463d84f278c45f3dba214d575667bcb5be9e7e6ecb","observation_id":"d685115a-385d-4405-b716-2dc4a66975b5","resolution":{"observed_at":"2026-07-04T09:39:46.911736Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"cited_work":{"arxiv_id":"2506.15483","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.15483","snapshot_observed_at":"2026-07-04T09:39:46.910380Z","title":"Task-oriented human- object interactions generation with implicit neural representations","venue":null,"work_id":"886fb306-daad-46e3-9245-7a3020fd7314","year":2025},"citing_paper":{"arxiv_id":"2606.28215","last_updated":"2026-06-26T16:05:58Z","snapshot_observed_at":"2026-08-07T08:50:21.361337Z","submitted_at":"2026-06-26T16:05:58Z","title":"HAT-4D: Lifting Monocular Video for 4D Multi-Object Interactions via Human-Agent Collaboration","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-29T04:18:02.341742Z"},"links":{"cited_paper":"/paper/2506.15483","citing_paper":"/paper/2606.28215"},"observation_digest":"sha256:b16bb67e560e070579dd33e4e1c25f40ed3a97b5e97c33a3d2b9833d321d9587","observation_id":"2f28fe85-7fee-4423-b8c4-a295a0a4703e","resolution":{"observed_at":"2026-07-01T17:05:50.601459Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.15483/citation-record","integrity":"/paper/2506.15483/integrity","json":"/paper/2506.15483/citation-record.json","paper":"/paper/2506.15483"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:55.480912Z","title":"Behave: Dataset and method for tracking human object interactions","venue":null,"work_id":"3a47dbe5-9022-4088-bffe-9e7528afe23f","year":2022},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:42.002765Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:a5309464233f88989d0c34b0a13acc93e778f9fd2fd50e91a485dceef899f689","observation_id":"8bb89a2a-4470-4081-8065-abdc7807f47c","resolution":{"observed_at":"2026-08-06T23:58:55.610618Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:55.234785Z","title":"Text2hoi: Text-guided 3d motion generation for hand-object interaction","venue":null,"work_id":"ed57668f-5d92-4c4b-a5f6-8e313e5b2c6e","year":2024},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:42.096650Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:27695d5efef91f58c89c176de642faabb77a8b83b7bad36bc019c8accd695ff7","observation_id":"cf8b1e63-adba-4c43-8c9d-71b636943c0e","resolution":{"observed_at":"2026-08-06T23:58:55.353068Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01291","last_updated":"2025-03-03T08:28:40Z","snapshot_observed_at":"2026-08-07T17:34:29.319657Z","submitted_at":"2025-03-03T08:28:40Z","title":"SemGeoMo: Dynamic Contextual Human Motion Generation with Semantic and Geometric Guidance","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01291","snapshot_observed_at":"2026-08-06T23:58:42.146744Z","title":"Semgeomo: Dynamic contextual hu- man motion generation with semantic and geometric guidance.arXiv preprint arXiv:2503.01291, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:42.146744Z"},"links":{"cited_paper":"/paper/2503.01291","citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:59e1dc47165e25fcbdda5996714d8900a8ee56b1f82c346bbba9df5417f83a3c","observation_id":"85e99873-00ef-4e60-ba51-2766fb97560d","resolution":{"observed_at":"2026-08-06T23:58:42.146744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:42.280842Z","title":"Human-object interaction with vision-language model guided relative movement dynamics","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:42.280842Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:f51670ebd709468bfae8f347ec3e9c824bdcf33d0266ca1ea2552e96e7d0fa27","observation_id":"b9d195ff-beb5-49aa-a07d-b54d14f5bd41","resolution":{"observed_at":"2026-08-06T23:58:42.280842Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:55.051818Z","title":"Cg-hoi: Contact-guided 3d human-object interaction genera- tion","venue":null,"work_id":"23c4cb9b-17c4-498c-ac53-b7155ad2af50","year":2024},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:42.413785Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:910af65edbabf3f71efe08d4676ef7f945637b2b0588196a12a5a9b6f8731055","observation_id":"9f4a522b-7c4f-4ae8-b817-7fe55c820f53","resolution":{"observed_at":"2026-08-06T23:58:55.130817Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:42.518312Z","title":"Arctic: A dataset for dexterous bimanual hand-object manipulation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:42.518312Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:7928e7b698966772539ad15edc1fc17f193292374bf015f4abc5920a4907bb1d","observation_id":"3fcb5930-6672-4c86-8c56-9cd9ada8aee2","resolution":{"observed_at":"2026-08-06T23:58:42.518312Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:42.617886Z","title":"3d-future: 3d furniture shape with texture","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:42.617886Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:40efca3dc55197927e3d28bd1277348b73a2e313d6b9a1b8ac2d571423affab9","observation_id":"05febdda-f983-40b5-bd5e-68b04c6d02ae","resolution":{"observed_at":"2026-08-06T23:58:42.617886Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:54.890420Z","title":"Coohoi: Learning cooperative human-object interaction with manipulated object dynamics","venue":null,"work_id":"6e7e604a-236e-4629-8f23-14764856f783","year":2024},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:42.740041Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:82692ef6fe85d8a4e1cabaf6b991bd54da1c5472877c6716585b7d189026997f","observation_id":"7b00bdab-6a56-4c82-af4f-8367bfc1c1f9","resolution":{"observed_at":"2026-08-06T23:58:54.955569Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:54.643100Z","title":"Auto-regressive diffusion for generating 3d human-object interactions","venue":null,"work_id":"12dac137-c3da-4959-9e33-ed87209324e6","year":2025},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:42.857804Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:e4e2c02a98ca394afc6b5beddbf649cf26a808f8c3151da571ad42260f2e005f","observation_id":"27d1151e-06cd-44fd-b7d3-a2a3b64e6a04","resolution":{"observed_at":"2026-08-06T23:58:54.784352Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:54.452817Z","title":"Imos: Intent-driven full-body motion synthesis for human-object interactions","venue":null,"work_id":"77c79de5-0cba-4fcb-baca-ff89f8d7dc7c","year":2023},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:42.981739Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:0e30f4fdc9914dcbffa9af1549c7098a489832d61f83e6a1aa563cbb4db747c5","observation_id":"cddcc914-3e35-4084-95ca-52ac5d239aed","resolution":{"observed_at":"2026-08-06T23:58:54.535978Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:54.197988Z","title":"Generating diverse and natural 3d human motions from text","venue":null,"work_id":"c60fb35b-30b1-4399-b38f-80f03da14d33","year":2022},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:43.117169Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:a07aaab644421fb899dfb518b01d7170d27791db891524d1e3f321387a8e5a5e","observation_id":"d41e7f0d-488a-4f5d-93da-11a84bd01880","resolution":{"observed_at":"2026-08-06T23:58:54.285016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06188","last_updated":"2025-05-09T17:25:34Z","snapshot_observed_at":"2026-07-06T18:43:14.028544Z","submitted_at":"2024-07-08T17:59:36Z","title":"CrowdMoGen: Zero-Shot Text-Driven Collective Motion Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.06188","snapshot_observed_at":"2026-08-06T23:58:43.178155Z","title":"Crowdmogen: Zero-shot text-driven collective motion generation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:43.178155Z"},"links":{"cited_paper":"/paper/2407.06188","citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:1ea59d59b163ae0579a3ccbfd990864a1b4fa37ae314b444e278fb125b9dd79e","observation_id":"8677cafd-01c2-46cf-a19f-f9085de28b16","resolution":{"observed_at":"2026-08-06T23:58:43.178155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:53.945099Z","title":"Stochastic scene-aware motion prediction","venue":null,"work_id":"ca68ddc0-c448-498f-9c23-41898fea7136","year":2021},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:43.243697Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:ee3bd16e286761abef2d41c196e718a9e0b13641f8e79aec4284196b1545efcb","observation_id":"d89d6b14-23ce-4a28-9cc2-b7bf1e2b18b6","resolution":{"observed_at":"2026-08-06T23:58:54.080788Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:53.680850Z","title":"Resolving 3d human pose ambiguities with 3d scene constraints","venue":null,"work_id":"a6efba66-c6c6-48ae-a383-dd5fe0e1e618","year":2019},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:43.295297Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:c0f98fcfaa2fad3642c6b792a585e13ce5b1c5df9a2d2aa750a18a1b4db50d8c","observation_id":"2db44fbe-bf52-4b32-95a1-ff7a1f0ed0d8","resolution":{"observed_at":"2026-08-06T23:58:53.767052Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:53.489310Z","title":"Nemf: Neural motion fields for kinematic animation","venue":null,"work_id":"f1f18322-eb29-4778-b0ee-8160953ab076","year":2022},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:43.377972Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:ba6b02b10a9719dd0fd40d5bb3e646b5a13df0bd98ee5fcf0ba02accfff08b26","observation_id":"8b84c955-aa78-4ce2-b618-49fca0fad081","resolution":{"observed_at":"2026-08-06T23:58:53.594127Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:43.450104Z","title":"Denoising diffusion probabilistic models","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:43.450104Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:7fef2763d7030b8a34d8cf5a3afd6d766359731a3affca745b1adfbde9fef2dc","observation_id":"778e589c-1164-47db-9272-658214621178","resolution":{"observed_at":"2026-08-06T23:58:43.450104Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:53.335629Z","title":"Diffusion-based generation, optimization, and planning in 3d scenes","venue":null,"work_id":"c007a1ea-2f06-43be-9062-2c2a054884b1","year":2023},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:43.509467Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:744162fba42395a10af4d6582c4486289d0e6eaff673c7a9f3a250056d732b08","observation_id":"ec09d9a2-c466-4b4e-8df2-79a11d1514bb","resolution":{"observed_at":"2026-08-06T23:58:53.403773Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:53.121256Z","title":"Intercap: Joint markerless 3d tracking of humans and objects in interaction","venue":null,"work_id":"4e314c33-7900-4c38-a149-653d20359194","year":2022},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:43.613275Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:8a2d806c04af3555af6df58491e33c8fdbadf08600954d55afe2743d28194b0e","observation_id":"aca699ab-8897-4745-a08a-38de29cdaf2c","resolution":{"observed_at":"2026-08-06T23:58:53.189998Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:52.952801Z","title":"Full-body articulated human-object interaction","venue":null,"work_id":"3f79f7c5-d49f-4688-8668-67aa5cc135e4","year":2023},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:43.712282Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:17d7c4db8b0cd2907d71caadbcd9033dbc0d315e115d86c1acb2c56d7f851575","observation_id":"8273493e-e640-4468-8a39-dc4c22ee0aa3","resolution":{"observed_at":"2026-08-06T23:58:53.041084Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08333","last_updated":"2025-08-11T11:45:30Z","snapshot_observed_at":"2026-08-06T18:36:43.603092Z","submitted_at":"2025-01-14T18:59:59Z","title":"DAViD: Modeling Dynamic Affordance of 3D Objects Using Pre-trained Video Diffusion Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08333","snapshot_observed_at":"2026-08-06T23:58:43.828049Z","title":"David: Modeling dynamic affordance of 3d objects using pre-trained video diffusion models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:43.828049Z"},"links":{"cited_paper":"/paper/2501.08333","citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:cf718fa58715b0b9b3b3ae12edbfd174384c1cb7359ea9fac794209f036ee777","observation_id":"b0ae1742-aaad-4d24-970d-93668506cf04","resolution":{"observed_at":"2026-08-06T23:58:43.828049Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.18600","last_updated":"2025-03-21T16:17:28Z","snapshot_observed_at":"2026-07-06T20:12:54.710203Z","submitted_at":"2024-12-24T18:55:38Z","title":"ZeroHSI: Zero-Shot 4D Human-Scene Interaction by Video Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.18600","snapshot_observed_at":"2026-08-06T23:58:43.894378Z","title":"Zerohsi: Zero-shot 4d human-scene interaction by video generation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:43.894378Z"},"links":{"cited_paper":"/paper/2412.18600","citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:4db1a7c8ce68ee9f61ea7a4ac8cc096f6a07b9b4efd14cb828cbabbb52462018","observation_id":"4dade97a-83b4-4102-8f52-9fd0d3b347dd","resolution":{"observed_at":"2026-08-06T23:58:43.894378Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:52.834883Z","title":"Controllable human-object interaction synthesis","venue":null,"work_id":"a1d7b510-3a34-4546-8a6f-8068233089b6","year":2024},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:43.996546Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:8b846fbe5352e46634a2fd503d0b5af0890971fa307ad80ba0e3dfa2c3407981","observation_id":"e24f0525-b002-4026-8f61-c62222bba92d","resolution":{"observed_at":"2026-08-06T23:58:52.887739Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:52.579439Z","title":"Object motion guided human motion synthesis","venue":null,"work_id":"eaa4f3e8-018f-4abf-a737-6a28a56bf7e0","year":2023},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:44.468408Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:f12f3031a29a43976b1a4819265140e59b3939eaf2441e2a8e338d0b099d68e5","observation_id":"b76309c1-7e2f-4d7a-8331-0a9c4e8c6220","resolution":{"observed_at":"2026-08-06T23:58:52.738866Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:52.394032Z","title":"Intergen: Diffusion-based multi-human motion generation under complex interactions","venue":null,"work_id":"7e8b8e29-602a-47f9-9523-058e56a8b01e","year":2024},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:45.214411Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:7288a09aff6c4c02142772c51742e2bb95e0a4294dc3033e80c15a8b47c5d86c","observation_id":"761e0590-d64d-4f3f-9027-ed3379687b2c","resolution":{"observed_at":"2026-08-06T23:58:52.468286Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:52.164233Z","title":"Learning basketball dribbling skills using trajectory optimization and deep reinforcement learning","venue":null,"work_id":"955c45ae-f6d0-480e-8bc8-c14b8385685a","year":2018},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:45.311591Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:f9f716c0808217e2ccbbf0e76f80ac29fcccd8095849af2d65f6862182a52c34","observation_id":"69c81e82-d443-4cb7-a092-61be08398c86","resolution":{"observed_at":"2026-08-06T23:58:52.291512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:51.988861Z","title":"Motion-x: A large-scale 3d expressive whole-body human motion dataset","venue":null,"work_id":"7cf00006-f6df-4ef2-8555-585bfa28ec6a","year":2023},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:45.390616Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:e93983e2f30dac7f9f6b3310f6916b61d2e7d55f523647ff01b3a988dc6d7031","observation_id":"ec0f6bfa-dac6-4331-a1fe-c1738896149f","resolution":{"observed_at":"2026-08-06T23:58:52.058585Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:51.871076Z","title":"Himo: A new benchmark for full-body human interacting with multiple objects","venue":null,"work_id":"6bdbe983-719e-4ec5-9c78-1416dbfcaf4f","year":2024},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:45.551315Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:510d1474e3dbdf2e1df724dbbec600df4360b1bc4bf5782018e0245f9cc452bc","observation_id":"d18afae9-064b-4a91-8855-275f85eaeecc","resolution":{"observed_at":"2026-08-06T23:58:51.922300Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:45.635299Z","title":"Amass: Archive of motion capture as surface shapes","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:45.635299Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:d90317e63de24d7c88f2d6406c2b6de8eba3b07b23da03ba5f4871335414b432","observation_id":"6b622f35-1ffd-4bd6-93be-0fa579c47152","resolution":{"observed_at":"2026-08-06T23:58:45.635299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:45.706998Z","title":"Expressive body capture: 3d hands, face, and body from a single image","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:45.706998Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:b2667fb93a6119d0ee6676ab689306b5cbf7c61575d8cf98d3896321ad3ac051","observation_id":"2237fa89-5080-4be3-bb9c-ca82cbde5b1d","resolution":{"observed_at":"2026-08-06T23:58:45.706998Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06553","last_updated":"2025-07-07T05:09:32Z","snapshot_observed_at":"2026-08-06T14:47:15.286683Z","submitted_at":"2023-12-11T17:41:17Z","title":"HOI-Diff: Text-Driven Synthesis of 3D Human-Object Interactions using Diffusion Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.06553","snapshot_observed_at":"2026-08-06T23:58:45.792596Z","title":"Hoi-diff: Text-driven synthesis of 3d human-object interactions using diffusion models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:45.792596Z"},"links":{"cited_paper":"/paper/2312.06553","citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:f5f48d7eeedd8fc8ccaebd9d2ba03afc9689ec30d9c8b7bea0bf1d01bf496eac","observation_id":"31686fc4-93d8-4b42-87e7-0e45c9473103","resolution":{"observed_at":"2026-08-06T23:58:45.792596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:51.683730Z","title":"Deepmimic: Example- guided deep reinforcement learning of physics-based character skills","venue":null,"work_id":"679962ac-8ebb-4382-a0aa-0b1b2a1ac510","year":2018},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:45.860094Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:27c4f75e827e050196da2a33fbae406d72ffe02bbde3994f2b57afcaf019cca9","observation_id":"ddf6243b-079a-4509-9228-7b4cd588e7c0","resolution":{"observed_at":"2026-08-06T23:58:51.783146Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:45.947221Z","title":"Amp: Adversarial motion priors for stylized physics-based character control","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:45.947221Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:8decda32ec27e974daf6f59d1fe1cd8d1a2a0ff9469eb262a60368f3619ac289","observation_id":"f0441dec-e8db-44b1-b207-d67e72be647e","resolution":{"observed_at":"2026-08-06T23:58:45.947221Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:51.453636Z","title":null,"venue":null,"work_id":"cdab25e0-2e8e-4512-92a9-a1b0f05ec825","year":2023},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:46.027688Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:0f9d90f3521d2b6b891774f239b83b0b736a54afa85c47960f22bd548f8451ea","observation_id":"9d5a696d-fdde-4695-be08-3e2489242943","resolution":{"observed_at":"2026-08-06T23:58:51.555638Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:46.235580Z","title":"Pointnet++: Deep hierarchical feature learning on point sets in a metric space","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:46.235580Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:8cae1970cbba599926b6469d01e56372b38e4897ba6830ab471276e81042f228","observation_id":"e44631c2-c95a-4c59-b02c-7ec88d863f98","resolution":{"observed_at":"2026-08-06T23:58:46.235580Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:46.339867Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:46.339867Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:a8b12584a9d0fc343b43eb46e5ab5c572a839679e1f445572da95b200cbd58e4","observation_id":"81d9f47f-e7df-496a-8f97-b85bb41a3f67","resolution":{"observed_at":"2026-08-06T23:58:46.339867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:51.306395Z","title":"Hoianimator: Generating text-prompt human-object animations using novel perceptive diffusion models","venue":null,"work_id":"dc684ed0-1b91-42b1-a2d4-3ccf5f473e98","year":2024},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:46.422003Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:69c2057fa4db3eb913a7aca89aefe2397ca4cf02b3847e95fc0fe24b11c58482","observation_id":"ddfc4ec0-aed6-4b92-b569-fb4a694e7267","resolution":{"observed_at":"2026-08-06T23:58:51.361816Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:46.501297Z","title":"A survey on human interaction motion generation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:46.501297Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:f5b4a7f401627ca53d0f54454b3189c5cc51a8daba0b49a38a94331e10d43caa","observation_id":"69e86349-d96d-4a6e-9f02-a34d0a55d574","resolution":{"observed_at":"2026-08-06T23:58:46.501297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:46.569130Z","title":"Grab: A dataset of whole-body human grasping of objects","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:46.569130Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:2875453ea9d526e2aa0e0776f1e80000bccb62dfff01983b4405a01af6a37316","observation_id":"ebfd29de-58b8-439a-b51a-4fc1b7828270","resolution":{"observed_at":"2026-08-06T23:58:46.569130Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:51.087817Z","title":"Human motion diffusion model","venue":null,"work_id":"b191eedf-b073-44b5-ade0-f11b7ab17014","year":2023},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:46.654005Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:ca79fec5352ba7d8248052ab650a55ce03b659823b3031d0739c758427ffdbca","observation_id":"27b22746-7cdd-4a96-8c9b-3b0206e71a5a","resolution":{"observed_at":"2026-08-06T23:58:51.194489Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04393","last_updated":"2023-12-07T16:06:31Z","snapshot_observed_at":"2026-08-06T12:06:41.391790Z","submitted_at":"2023-12-07T16:06:31Z","title":"PhysHOI: Physics-Based Imitation of Dynamic Human-Object Interaction","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04393","snapshot_observed_at":"2026-08-06T23:58:46.767124Z","title":"Physhoi: Physics-based imitation of dynamic human-object interaction","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:46.767124Z"},"links":{"cited_paper":"/paper/2312.04393","citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:320c9d81f95a509ac93dec003bdaf42c79a3302d78daba0d2e410f67c1a23ef4","observation_id":"e943a2a1-1745-454c-a6ab-e1fac9ac3ad4","resolution":{"observed_at":"2026-08-06T23:58:46.767124Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:50.873507Z","title":"Move as you say interact as you can: Language-guided human motion generation with scene affordance","venue":null,"work_id":"ca962f40-3d7e-4e73-b6bf-7081cc9ec50a","year":2024},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:46.836988Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:f4d05c30e2f1561ef4b2b42d46d602ad80ee8ccb28a5676aba8b11bd1a5f9409","observation_id":"81684676-5982-45ac-82fb-997deb5a267b","resolution":{"observed_at":"2026-08-06T23:58:50.974359Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:50.635286Z","title":"Humanise: Language-conditioned human motion generation in 3d scenes","venue":null,"work_id":"32fbec90-8e75-4d13-aa81-e12963364567","year":2022},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:46.949651Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:8215946a65a1770a552cf1aead9ea9f93b1aadfac27378a587a56fb7b9487ca3","observation_id":"bbff33ce-185e-4677-a112-366ae08ab09c","resolution":{"observed_at":"2026-08-06T23:58:50.772239Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.15898","last_updated":"2025-03-20T06:50:18Z","snapshot_observed_at":"2026-08-07T16:49:57.551803Z","submitted_at":"2025-03-20T06:50:18Z","title":"Reconstructing In-the-Wild Open-Vocabulary Human-Object Interactions","version":1},"cited_work":{"arxiv_id":"2503.15898","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.15898","snapshot_observed_at":"2026-08-06T23:58:48.756726Z","title":"Reconstructing In-the-Wild Open-Vocabulary Human-Object Interactions","venue":"cs.CV","work_id":"89abc6b2-1fdc-49d1-b6cb-474788fffb3f","year":2025},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:47.018564Z"},"links":{"cited_paper":"/paper/2503.15898","citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:f6da49019a7f045c9c5960f9c0f827cba33fe574d6117633bc2d7367acd6030f","observation_id":"15d450c7-0e08-4df3-84ef-3cb8fb5baa37","resolution":{"observed_at":"2026-08-06T23:58:48.860854Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11208","last_updated":"2024-03-17T13:17:25Z","snapshot_observed_at":"2026-08-06T13:04:39.044059Z","submitted_at":"2024-03-17T13:17:25Z","title":"THOR: Text to Human-Object Interaction Diffusion via Relation Intervention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11208","snapshot_observed_at":"2026-08-06T23:58:47.146503Z","title":"Thor: Text to human-object interaction diffusion via relation intervention","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:47.146503Z"},"links":{"cited_paper":"/paper/2403.11208","citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:b0a9104689e2b9badde496d4f2153bbcfa74b1a5fa50122be76cfc71762457b3","observation_id":"e04b1c83-bc29-4261-8147-14d46a3cbd63","resolution":{"observed_at":"2026-08-06T23:58:47.146503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17840","last_updated":"2025-08-21T08:23:55Z","snapshot_observed_at":"2026-07-06T18:36:56.124021Z","submitted_at":"2024-06-25T17:46:28Z","title":"Human-Object Interaction from Human-Level Instructions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17840","snapshot_observed_at":"2026-08-06T23:58:47.255699Z","title":"Human-object interaction from human-level instructions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:47.255699Z"},"links":{"cited_paper":"/paper/2406.17840","citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:5f8b55e96dd4fe3f52cbbc0578e8d7e5233b5832b12ec62614829817f066ad91","observation_id":"0f7732d0-78c0-4d75-a9ef-811590ab6eea","resolution":{"observed_at":"2026-08-06T23:58:47.255699Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:50.302520Z","title":"Inter-x: Towards versatile human-human interac- tion analysis","venue":null,"work_id":"e08bcefe-cf22-4d0b-a2fc-b88443187e15","year":2024},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:47.324819Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:c29f633bcd5f792a4d33a9810c2d0d9d2409f4266faf98aaee53fa5b4222fb6f","observation_id":"f07d24a9-71a9-4ef3-a05b-5df159e40401","resolution":{"observed_at":"2026-08-06T23:58:50.467461Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:47.416700Z","title":"Interdiff: Generating 3d human-object interactions with physics-informed diffusion","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:47.416700Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:5644955b1fd321580161fa2f048fcfe7b6913933c881d2dbc41e26b1ad389e5d","observation_id":"fc17981d-cd17-4caf-b811-aa47b6550f6e","resolution":{"observed_at":"2026-08-06T23:58:47.416700Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:47.522418Z","title":"Intermimic: Towards universal whole-body control for physics-based human-object interactions","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:47.522418Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:e50d204fc3e1d4ef413e2a5e552974868b810c9f0e31af48c34bcb71f9c2994e","observation_id":"0fb00f35-0c3e-45db-8298-9f68ec685030","resolution":{"observed_at":"2026-08-06T23:58:47.522418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:47.624950Z","title":"Interdreamer: Zero-shot text to 3d dynamic human-object interaction","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:47.624950Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:32434942717adfddb7018b8cf27a776dc6736271e04e6df48a17dcca87663afb","observation_id":"577b72e0-40c9-4ea6-9920-caa88a87fb22","resolution":{"observed_at":"2026-08-06T23:58:47.624950Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:50.073823Z","title":"F-hoi: Toward fine- grained semantic-aligned 3d human-object interactions, 2024","venue":null,"work_id":"c3dc3b0e-562c-418b-bfd9-848f04b7e73b","year":2024},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:47.757343Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:1e5587fa4b9447e6480d95afbfc0da77d06bd79db0bbabfd54df8b82e376c1de","observation_id":"d804603e-af5e-4944-b6bd-b0195460fa42","resolution":{"observed_at":"2026-08-06T23:58:50.171636Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:49.796406Z","title":"Generating human interaction motions in scenes with text control","venue":null,"work_id":"10f06c59-9e6a-40a4-8c3a-79e55af476fe","year":2024},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:47.846996Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:71215ae3bb7e0d975a4cf7df62254ad6d936dc7b1897f14c29edc2362f96753e","observation_id":"aab35056-7b66-4679-850e-2ddf317a14c8","resolution":{"observed_at":"2026-08-06T23:58:49.923478Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.13130","last_updated":"2025-03-17T12:55:34Z","snapshot_observed_at":"2026-08-07T16:58:04.625418Z","submitted_at":"2025-03-17T12:55:34Z","title":"ChainHOI: Joint-based Kinematic Chain Modeling for Human-Object Interaction Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.13130","snapshot_observed_at":"2026-08-06T23:58:47.939099Z","title":"Chainhoi: Joint-based kinematic chain modeling for human-object interaction generation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:47.939099Z"},"links":{"cited_paper":"/paper/2503.13130","citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:23a300d1fdfce55296dd383d9b24b439b248d66edfb3067dd0bcf78a73232f6a","observation_id":"6017ce03-d665-4fa8-9eb8-de93d2096b17","resolution":{"observed_at":"2026-08-06T23:58:47.939099Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.19353","last_updated":"2024-12-24T09:33:24Z","snapshot_observed_at":"2026-08-07T09:29:39.191971Z","submitted_at":"2024-06-27T17:32:18Z","title":"CORE4D: A 4D Human-Object-Human Interaction Dataset for Collaborative Object REarrangement","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.19353","snapshot_observed_at":"2026-08-06T23:58:48.047771Z","title":"Core4d: A 4d human- object-human interaction dataset for collaborative object rearrangement","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:48.047771Z"},"links":{"cited_paper":"/paper/2406.19353","citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:8a254be034cefb1638362cf90ce470c8dba4d22e110f3aa85a9ca4d0860661d5","observation_id":"e8fe0b82-aab9-4a9d-9ad1-a6b1fe69443d","resolution":{"observed_at":"2026-08-06T23:58:48.047771Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:49.462939Z","title":"Couch: Towards controllable human-chair interactions","venue":null,"work_id":"5eba9a11-1bde-41e2-b7eb-86747b10d624","year":2022},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:48.200420Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:42b9c0b28c2e926773b4908f01f30ea2851cf704e0566c4d7a6e7863ad5c3df1","observation_id":"4ae07036-8fec-41fe-95b4-ff15fcf28ac9","resolution":{"observed_at":"2026-08-06T23:58:49.635573Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:58:49.193097Z","title":"On the continuity of rotation representations in neural networks","venue":null,"work_id":"bea441c5-89ed-4f05-90dd-5f39d1b8ef79","year":2019},"citing_paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T23:58:48.341938Z"},"links":{"citing_paper":"/paper/2506.15483"},"observation_digest":"sha256:ae56f91a592b32c5aa9993d4a7497cd9a27ffaf99da82a36167b47e9eec577aa","observation_id":"e9a529ed-d606-4a58-8b38-ed3e9a39a955","resolution":{"observed_at":"2026-08-06T23:58:49.333983Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.15483","last_updated":"2025-06-18T14:17:53Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T09:30:58.805804Z","submitted_at":"2025-06-18T14:17:53Z","title":"GenHOI: Generalizing Text-driven 4D Human-Object Interaction Synthesis for Unseen Objects"},"reference_resolution":{"displayed":55,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":25,"verified_exact":1,"verified_fuzzy":29},"total_outbound_references":55},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 55 of 55 outbound references and 4 inbound Pith citation observations for arXiv:2506.15483."}