{"as_of":"2026-08-19T09:11:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b5144bb620b6a40c4dc644f0cc406d524454b955f1a7ba0f003c79a85e2615c8","coverage":[{"denominator":117,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T17:55:59.446190Z","state":"measured"},{"denominator":100,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":100,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.13049/citation-record","integrity":"/paper/2608.13049/integrity","json":"/paper/2608.13049/citation-record.json","paper":"/paper/2608.13049"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:58.984736Z","title":"OpenAI Blog , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:58.984736Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:49684d8455b9e8cc85a288404eae5dcf343bc90728c41ff7210eaeba174ecc4c","observation_id":"94143abd-01cb-44ae-ab46-af8c1c6a0d91","resolution":{"observed_at":"2026-08-15T17:55:58.984736Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:58.988938Z","title":"Forty-first International Conference on Machine Learning , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:58.988938Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:0f60c49444790a157fad2dc95fb3ab376d8f6b6cb6b717293cde26f5545544a4","observation_id":"945babf7-1e28-4b82-ab86-47a48c5d27f0","resolution":{"observed_at":"2026-08-15T17:55:58.988938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:58.992405Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:58.992405Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:e7e5985e43bfddac3ddffae856942d3025df1a87072cf72d567ad99207f874a5","observation_id":"494226dd-fc61-4b2b-a26c-be9d876ba8ae","resolution":{"observed_at":"2026-08-15T17:55:58.992405Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.003285Z","title":"Conference on Robot Learning , pages=","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.003285Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:0e463f6d5055b78a3aff86c14d60651c0a990c34da33cf67c81842cb39bbcf86","observation_id":"c5cf053c-803d-47ea-a50f-82ede72f6ca4","resolution":{"observed_at":"2026-08-15T17:55:59.003285Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.007014Z","title":"2024 IEEE International Conference on Robotics and Automation (ICRA) , pages=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.007014Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:f3a3d67564f0b89393dd832d804c9d28beb0457cf1b2a8452d782579c0a41533","observation_id":"39221a9c-74ad-4dff-a1d3-2a0d3554d5c1","resolution":{"observed_at":"2026-08-15T17:55:59.007014Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.021614Z","title":"International Journal of Computer Vision , volume=","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.021614Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:2077b6129dabfa7a7989c20f10c77033a7e300edc207bee7b1fb360083c4707b","observation_id":"9d98b71e-5f3e-4f0a-9210-86c7d928158a","resolution":{"observed_at":"2026-08-15T17:55:59.021614Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.025263Z","title":"Proceedings of the IEEE/CVF conference on computer vision and pattern recognition , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.025263Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:336cbf00168d7b1bc86b9a5403adca9226eba53f6afee132fa284709d51434d9","observation_id":"d336fd52-5ca5-43cb-8235-9b7b30b961a2","resolution":{"observed_at":"2026-08-15T17:55:59.025263Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.051655Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.051655Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:a7be1c5d3e65c462655f5d06bdea459298d7425fffac0b563ff4611b8d062e49","observation_id":"36b7d027-d9c1-4cd3-be8c-c4ce8f91a386","resolution":{"observed_at":"2026-08-15T17:55:59.051655Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.060495Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.060495Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:5717d6a5eafe02b70436f06289845558b8ab72fa7363673d5d3fa47410f05523","observation_id":"a7a5ff3f-9b6c-43ec-b673-9959474d543b","resolution":{"observed_at":"2026-08-15T17:55:59.060495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.069548Z","title":"IEEE Transactions on human-machine systems , volume=","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.069548Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:bc2d5ef3f99c0af2ff48bca1ab854b2542f0af3cc8669d65b336dae5402be4e6","observation_id":"5587a01a-7343-48e4-9d1e-daed0f8e287a","resolution":{"observed_at":"2026-08-15T17:55:59.069548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.072967Z","title":"Frontiers in Neurorobotics , volume=","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.072967Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:4cbfdd293821bfda1cf65c9caea792c5283346306f1745758672e8abc134f934","observation_id":"318dce59-b226-4d9c-ad4f-6328b574dc4e","resolution":{"observed_at":"2026-08-15T17:55:59.072967Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.12609","last_updated":"2025-08-16T04:32:53Z","snapshot_observed_at":"2026-08-17T19:18:34.104433Z","submitted_at":"2025-04-17T03:15:20Z","title":"Crossing the Human-Robot Embodiment Gap with Sim-to-Real RL using One Human Demonstration","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.12609","snapshot_observed_at":"2026-08-15T17:55:59.076620Z","title":"arXiv preprint arXiv:2504.12609 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.076620Z"},"links":{"cited_paper":"/paper/2504.12609","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:113864f68b5c66779ce657d7eb91c9ea6d78979fba1cf1d6814d74507b776df7","observation_id":"49f0b00d-9db5-4795-a60e-d4e1139d041c","resolution":{"observed_at":"2026-08-15T17:55:59.076620Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04169","last_updated":"2025-01-07T22:33:47Z","snapshot_observed_at":"2026-08-19T08:25:45.039825Z","submitted_at":"2025-01-07T22:33:47Z","title":"Learning to Transfer Human Hand Skills for Robot Manipulations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.04169","snapshot_observed_at":"2026-08-15T17:55:59.080225Z","title":"arXiv preprint arXiv:2501.04169 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.080225Z"},"links":{"cited_paper":"/paper/2501.04169","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:e6d37865d896782cc97b3fa304ab23b593acdfaa409139dd9bc1080263e6d4f9","observation_id":"bd9ac34d-e229-4455-9bf3-4a86340939f3","resolution":{"observed_at":"2026-08-15T17:55:59.080225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.14792","last_updated":"2022-09-29T13:59:46Z","snapshot_observed_at":"2026-07-06T13:57:47.051387Z","submitted_at":"2022-09-29T13:59:46Z","title":"Make-A-Video: Text-to-Video Generation without Text-Video Data","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.14792","snapshot_observed_at":"2026-08-15T17:55:59.083876Z","title":"arXiv preprint arXiv:2209.14792 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.083876Z"},"links":{"cited_paper":"/paper/2209.14792","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:ca58f2287629dda358e13b0b1a9565e723d055ab4a91c75aff3525c747083345","observation_id":"a7c70983-5d2c-47d2-8d10-a82301544d38","resolution":{"observed_at":"2026-08-15T17:55:59.083876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.02303","last_updated":"2022-10-05T14:41:38Z","snapshot_observed_at":"2026-08-18T00:41:06.039123Z","submitted_at":"2022-10-05T14:41:38Z","title":"Imagen Video: High Definition Video Generation with Diffusion Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.02303","snapshot_observed_at":"2026-08-15T17:55:59.088092Z","title":"arXiv preprint arXiv:2210.02303 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.088092Z"},"links":{"cited_paper":"/paper/2210.02303","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:2049e54bed46b303bddde3e17668a4eb1139cfde822303b6b70a0156c51e4200","observation_id":"448d490a-45f2-40bf-8f9b-d2cd1d01aac7","resolution":{"observed_at":"2026-08-15T17:55:59.088092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.02399","last_updated":"2022-10-05T17:18:28Z","snapshot_observed_at":"2026-08-15T13:11:17.068630Z","submitted_at":"2022-10-05T17:18:28Z","title":"Phenaki: Variable Length Video Generation From Open Domain Textual Description","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.02399","snapshot_observed_at":"2026-08-15T17:55:59.092365Z","title":"arXiv preprint arXiv:2210.02399 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.092365Z"},"links":{"cited_paper":"/paper/2210.02399","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:0553af2f71c90326723c0303bdee2efb9c7807da764776badb29b0b3cc2081f7","observation_id":"530b4fa5-0328-4e5b-b578-87556e7949ba","resolution":{"observed_at":"2026-08-15T17:55:59.092365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.096483Z","title":"Proceedings of the IEEE/CVF conference on computer vision and pattern recognition , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.096483Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:f20f08040870521e640f2163976e05716d36d8dad66abd2653e657d8ac4e9cf0","observation_id":"fb5f36d3-13f9-417f-80a5-19c8be87038b","resolution":{"observed_at":"2026-08-15T17:55:59.096483Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04725","last_updated":"2024-02-08T18:08:57Z","snapshot_observed_at":"2026-08-17T14:04:31.230742Z","submitted_at":"2023-07-10T17:34:16Z","title":"AnimateDiff: Animate Your Personalized Text-to-Image Diffusion Models without Specific Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.04725","snapshot_observed_at":"2026-08-15T17:55:59.100373Z","title":"arXiv preprint arXiv:2307.04725 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.100373Z"},"links":{"cited_paper":"/paper/2307.04725","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:dea77c9a7f2002059929826a31997c82bbc81ee77f90b8a9ca13edb4e77715ff","observation_id":"f9d2c33c-44d5-44ce-b40f-c2b8fa10ac2b","resolution":{"observed_at":"2026-08-15T17:55:59.100373Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.104745Z","title":"Proceedings of the IEEE/CVF International Conference on Computer Vision , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.104745Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:ee855a40e486fc143165e2645566da9345f84aedbec187b2626fee01ec6b4b48","observation_id":"4fe8a98a-c918-407f-a108-e9aab3590936","resolution":{"observed_at":"2026-08-15T17:55:59.104745Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.108026Z","title":"Proceedings of the IEEE/CVF conference on computer vision and pattern recognition , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.108026Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:2e056d75d7d3466ef1be9c7527f9919506a86f4999b5b5f58ceb68cfcdd4b15a","observation_id":"3f82a80c-fdc3-42f0-915b-3583c32787c0","resolution":{"observed_at":"2026-08-15T17:55:59.108026Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.111147Z","title":"SIGGRAPH Asia 2024 Conference Papers , pages=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.111147Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:40724d374db985766eed29db750185ccbd081e1dce02893a7359175946fbb7ac","observation_id":"68922ce3-b2be-4a29-8b4d-ec22e2ca9319","resolution":{"observed_at":"2026-08-15T17:55:59.111147Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.114343Z","title":"International Conference on Learning Representations , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.114343Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:04e94bac666bc7ef9b4f654779823f93e67e41667c6ae8f3383c00bc2fe0aa32","observation_id":"bfe5b6b2-e740-483a-bdfd-93cd399ed981","resolution":{"observed_at":"2026-08-15T17:55:59.114343Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.132484Z","title":"International Conference on Machine Learning , pages=","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.132484Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:aa2ae29319ed47595e8e580e3ff723b2b2d538308068c2d68a7f09cd3d3df522","observation_id":"d203c69f-3d76-44a1-82c2-cf9e9f013b3a","resolution":{"observed_at":"2026-08-15T17:55:59.132484Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.136362Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.136362Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:40ae77b4db8009e4b2dd03da612a1a931b6963676289e643ab7b1f060db1f4a9","observation_id":"1c941f83-9495-4a8d-ae23-d59dd5fa9abf","resolution":{"observed_at":"2026-08-15T17:55:59.136362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.139584Z","title":"European Conference on Computer Vision , pages=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.139584Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:5d29bacddc7c363cb4c6d7692887593e522b7a912c74b74e646b61217ad455b9","observation_id":"4bdab6bd-4888-4843-a865-ab8fb8f2e424","resolution":{"observed_at":"2026-08-15T17:55:59.139584Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.142784Z","title":"Proceedings of the Computer Vision and Pattern Recognition Conference , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.142784Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:2b4dde7013f802bd950c3b68b9264927665d9cdf0a6f73f4279b08e37acc28eb","observation_id":"03e8e667-d3fa-4714-9219-bf09cefeebe1","resolution":{"observed_at":"2026-08-15T17:55:59.142784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.145740Z","title":"International Conference on Learning Representations , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.145740Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:6631145c455475664dc26ba9d2e5a9d0e396787709e6b7973108778d07f6b6f1","observation_id":"63275f5a-3868-4b59-bee9-6131f03180a9","resolution":{"observed_at":"2026-08-15T17:55:59.145740Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.149082Z","title":"Proceedings of the IEEE/CVF international conference on computer vision , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.149082Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:de4b29d95956eb388b82e3636b2250293bff4a80db73775d9ec5331a1824e775","observation_id":"974c853d-cb8d-48ba-b4a4-9d6894aec1c0","resolution":{"observed_at":"2026-08-15T17:55:59.149082Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.159301Z","title":"Proceedings of the IEEE/CVF conference on computer vision and pattern recognition , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.159301Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:c1efeb04d5147b4dd88f3c7adbcfddb625fb04e1df953d5e2794b388f9156b66","observation_id":"c20649bd-fbb6-484e-ad3d-8133e5a74b7e","resolution":{"observed_at":"2026-08-15T17:55:59.159301Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.162370Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.162370Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:be7ae32e75243bbacca350863ae508aa262507f371cb8fc0ff99e15218d7c0f4","observation_id":"31ac8e0a-101f-419f-a209-10a38c4fb53d","resolution":{"observed_at":"2026-08-15T17:55:59.162370Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.165405Z","title":"International Conference on Learning Representations , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.165405Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:436f8e5cd4f6f51368d2e73fc00a695b8fa4136a4743b055fd12da3d732aee75","observation_id":"bebed815-59e7-4712-ac25-a0851992f4b6","resolution":{"observed_at":"2026-08-15T17:55:59.165405Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.169392Z","title":"Proceedings of the Computer Vision and Pattern Recognition Conference , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.169392Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:f682c565a3718ead1cbd0f64a280e66ec02ea12969fbc47cbd7245bbdb9bf41b","observation_id":"cd8f0462-e272-4867-ba83-c0fe4863e155","resolution":{"observed_at":"2026-08-15T17:55:59.169392Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.172831Z","title":"Proceedings of the 64th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.172831Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:3027e3c84deb963a507c31b6463c93fafb6eaaefb9ac01bdf548f2399241493c","observation_id":"300e7c57-8f92-4e9e-a773-c19f25376b02","resolution":{"observed_at":"2026-08-15T17:55:59.172831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.176012Z","title":"Proceedings of the AAAI Conference on Artificial Intelligence , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.176012Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:394a10549bf0d177bfefc079049dddd8deb6e2e5166c1542d2109c34eccca5b2","observation_id":"170e35f6-4a1f-41d8-a767-33d7a11f8fd5","resolution":{"observed_at":"2026-08-15T17:55:59.176012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.190455Z","title":"2025 , month = oct, howpublished =","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.190455Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:9fee04649cee358c6f7da048aaddf615597cc19b5babf43df258c66c4c658bfd","observation_id":"99da4947-b069-4b88-8cd6-e17958bd4e11","resolution":{"observed_at":"2026-08-15T17:55:59.190455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.193671Z","title":"2026 , month = jan, note =","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.193671Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:a80d5a7572d39c9e7371d41097e4841372c504bc4caf62867836c28d2bcf2021","observation_id":"6e49f7ed-6b20-4ef5-a9e7-bf006b33bf39","resolution":{"observed_at":"2026-08-15T17:55:59.193671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.212579Z","title":"Proceedings of the IEEE/CVF international conference on computer vision , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.212579Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:88db9ac4f49dc422b9d90a45ff95d5e92de0531ef9888be1d142d0b21bafd21c","observation_id":"769349cf-c610-4ae7-a0b2-bdf6a865cfdd","resolution":{"observed_at":"2026-08-15T17:55:59.212579Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.215820Z","title":"International conference on machine learning , pages=","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.215820Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:0561dd07d938ae3386935c6c9de200d1a8d508a4ce42196f2acb3fd303f92ceb","observation_id":"88d301cf-5112-4429-9d72-7c7593884c08","resolution":{"observed_at":"2026-08-15T17:55:59.215820Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.218762Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.218762Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:5d5893fb254d8ef1ba645f9f0819f62146ddedffe114a0074aad8cf0c30a69a1","observation_id":"7f1197bb-526d-430c-96f8-00dd4b7b67c7","resolution":{"observed_at":"2026-08-15T17:55:59.218762Z","resolver_source":null,"status":"parse_uncertain"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.221857Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.221857Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:56b6b92f624b48da80da0cd689abfa528e6529bcdfa3c6b90d723eee1d9365ec","observation_id":"28f458b1-d112-42d6-8f1e-0915c9ea42fb","resolution":{"observed_at":"2026-08-15T17:55:59.221857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.237825Z","title":"Advances in Neural Information Processing Systems (NeurIPS) , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.237825Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:4bee65b3ac1336bc87442ac3ae2f3080106409ccda1b34a621be3b1d94685c62","observation_id":"8892f8e1-5e75-4b13-8615-2908d02df578","resolution":{"observed_at":"2026-08-15T17:55:59.237825Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.241787Z","title":"Proceedings of the International Conference on Machine Learning (ICML) , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.241787Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:4e4fbd13c1caab7c51961269340f084aee541029985ccf13f9dee7f876763d99","observation_id":"aeb915b6-90b3-4e85-a02e-9eab461e96ee","resolution":{"observed_at":"2026-08-15T17:55:59.241787Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.245675Z","title":"European Conference on Computer Vision (ECCV) , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.245675Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:f1154a6fbc4e8d1647054e99834186f8a1862e5c807c5727c50ba292e5ac4c0b","observation_id":"0bf79fc2-655f-48d1-b293-a881b9809ef7","resolution":{"observed_at":"2026-08-15T17:55:59.245675Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:01.068865Z","title":"Advances in Neural Information Processing Systems (NeurIPS) , year=","venue":null,"work_id":"5804b031-8e15-4629-b3ef-227b7b4b401d","year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.255961Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:1971084555f0dc339534625e65443fa45c9539ee0e3a6986da02393aa848bb2c","observation_id":"3a398c53-db05-4e50-ab8a-8d242dcd4abc","resolution":{"observed_at":"2026-08-15T17:56:01.072327Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:01.058994Z","title":"British Machine Vision Conference (BMVC) , year=","venue":null,"work_id":"56742d82-75da-4c52-8edd-d03936274291","year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.259452Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:1206829c808c1944f3afc310ab78fbc1ee82e5603977304f5bb9b14db2f77bb5","observation_id":"dc41e7c3-0a9d-4658-bafc-6d9f2a8f090c","resolution":{"observed_at":"2026-08-15T17:56:01.062219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:01.048715Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) , year=","venue":null,"work_id":"d30e8332-74bf-4c51-b1ff-bae8148c8ec5","year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.263397Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:abd6aec86109bd67d5c8682d8da9395f0da544917264f2be07872eea05532cdd","observation_id":"45040613-311c-4c35-8b3e-117296c816e9","resolution":{"observed_at":"2026-08-15T17:56:01.052055Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:01.038544Z","title":"International Journal of Computer Vision (IJCV) , year=","venue":null,"work_id":"7bc1cc94-94eb-4412-bf39-cf02acb55b97","year":null},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.266753Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:b6e10db2600d24627b271b3844ced90466ba31cffd9cf0ea1513ba3843a987bc","observation_id":"11cdf0f7-379d-4ab5-b35d-1539b509ef3c","resolution":{"observed_at":"2026-08-15T17:56:01.042462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:01.029032Z","title":"Displays , volume=","venue":null,"work_id":"ba70fd5c-318b-4450-b424-19a7240b23b8","year":2025},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.273438Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:643f2698f19259d23b68a86e33f6ddbb50b70d7f77141fd9f8c4dec8bf0a1050","observation_id":"6a2003cd-f9f3-426e-b5c4-0f5f3e170b57","resolution":{"observed_at":"2026-08-15T17:56:01.032470Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:01.019968Z","title":null,"venue":null,"work_id":"538672c5-1b5b-4f13-992c-e549b9c663f0","year":2023},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.277154Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:0ef952bfd7ad324304af6437c4f534251728663af5ad50c19df9cb55481172a4","observation_id":"73bf6d3e-5e5e-4c86-9875-b2b464a265aa","resolution":{"observed_at":"2026-08-15T17:56:01.023109Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:01.010079Z","title":null,"venue":null,"work_id":"15ae0d2a-7694-44bd-9d23-65b427badfd7","year":2025},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.281337Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:eaee78d966967e3773f9fd5a691622663153bfcbec92434954d1be6c77c681fd","observation_id":"c87b0337-d86b-4114-975f-5addb6c910e3","resolution":{"observed_at":"2026-08-15T17:56:01.013280Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.284714Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.284714Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:41a139f81ea4702fb0add11246a4d3351becc5f3b0db45c8a19160801edaa823","observation_id":"b7ec192e-1ff0-42d9-b165-c027cdae343f","resolution":{"observed_at":"2026-08-15T17:55:59.284714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.994253Z","title":null,"venue":null,"work_id":"4256a4ff-deb3-4e0d-9a2d-10fb1887dad3","year":2024},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.287871Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:51d4549d830e5df567dbb418ad446859e921b95a150c8070dcef0e55ad845604","observation_id":"2ff52672-2ebd-4e63-9857-15efb3b731de","resolution":{"observed_at":"2026-08-15T17:56:00.997823Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.24164","last_updated":"2026-01-08T17:01:05Z","snapshot_observed_at":"2026-08-16T17:53:54.636855Z","submitted_at":"2024-10-31T17:22:30Z","title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.24164","snapshot_observed_at":"2026-08-15T17:55:59.291015Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.291015Z"},"links":{"cited_paper":"/paper/2410.24164","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:44c7c26cb279cd7574a913eef49c9d0d4753b9d48c5370b3cb15e01fca9610a5","observation_id":"eb104190-ba30-44e6-870e-bab3fd288e9e","resolution":{"observed_at":"2026-08-15T17:55:59.291015Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.984462Z","title":"W.; Fidler, S.; and Kreis, K","venue":null,"work_id":"2df168a2-5677-4a26-b5b9-16cef77bbf8e","year":2023},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.294550Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:23af4f5e306409f0cae450d2015406496d6f85155e62448b3b1d444849467bea","observation_id":"13223cc5-249c-4649-9ce7-9b08f785b093","resolution":{"observed_at":"2026-08-15T17:56:00.987842Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.06817","last_updated":"2023-08-11T17:45:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-12-13T18:55:15Z","title":"RT-1: Robotics Transformer for Real-World Control at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.06817","snapshot_observed_at":"2026-08-15T17:55:59.298166Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.298166Z"},"links":{"cited_paper":"/paper/2212.06817","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:644ac310a4b6d231bc088e74afbd278f8639c86490be04ed13c62467667afa88","observation_id":"a98604d8-c397-47de-ad8e-889d9adee419","resolution":{"observed_at":"2026-08-15T17:55:59.298166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.974097Z","title":null,"venue":null,"work_id":"1b5df2fc-72d2-4b61-986c-46faa299b474","year":2024},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.301219Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:88f590cadbe29432651a68243ba493babde4e3944ebdb652ba40ca679039ac3d","observation_id":"1eaa451e-d33f-40d3-a255-32359c22385a","resolution":{"observed_at":"2026-08-15T17:56:00.977822Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.963134Z","title":"D.; Edwards, A.; Parker-Holder, J.; Shi, Y.; Hughes, E.; Lai, M.; Mavalankar, A.; Steigerwald, R.; Apps, C.; et al","venue":null,"work_id":"8b848957-43a3-492d-8c25-0c0d6a96be84","year":2024},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.304798Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:40997a14ed3b7cb457b2005ab992f0ff2c5cfc3ca2270f5843c2dc81ec2ced7c","observation_id":"34d893fa-745e-48ce-ab41-406b61a728c0","resolution":{"observed_at":"2026-08-15T17:56:00.967054Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.953683Z","title":null,"venue":null,"work_id":"c37dfc40-7c26-4b97-9658-4a8840c5f0ca","year":2025},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.308193Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:4d72db42c504b2dfe35a8702de3613d7484e9e07562c666499d4e0d5d61c842a","observation_id":"92490752-b270-4b18-b5dd-c1496dadd8c4","resolution":{"observed_at":"2026-08-15T17:56:00.957074Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.944242Z","title":null,"venue":null,"work_id":"d6d7fd4e-e603-4e29-aa53-eeca1ca42031","year":2025},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.311256Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:e29ae6b7280fff8ad8104d69a5a1f79574b5c7adc3d56a38048d39a800f19fb4","observation_id":"b5d9b506-5e95-487b-aa7d-ec264d7596a9","resolution":{"observed_at":"2026-08-15T17:56:00.947539Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.314406Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.314406Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:7e4b02eb6f8d13f28ad844b29f7adfd0b003fdf0207243a8b21351883c2e98fd","observation_id":"6fe1e778-0983-410a-8df3-3285eb74fb68","resolution":{"observed_at":"2026-08-15T17:55:59.314406Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.934280Z","title":"M.; Furnari, A.; Kazakos, E.; Ma, J.; Moltisanti, D.; Munro, J.; Perrett, T.; Price, W.; et al","venue":null,"work_id":"3422aa52-bd6e-4bd7-854c-140e62dd24cd","year":2022},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.317896Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:fff29f04d5e91e7b0113bf33c5f6f899ba3eacccb9bb2c9a5bef532c9b769265","observation_id":"d53278d3-7230-46a8-92fe-beae060f964e","resolution":{"observed_at":"2026-08-15T17:56:00.937427Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.321339Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.321339Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:56835c560c63d93547c4f20446e01f66015df6847a41b09af121976b77aeddab","observation_id":"2f6cab9b-a682-4126-9274-71a49e9536bb","resolution":{"observed_at":"2026-08-15T17:55:59.321339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.924643Z","title":null,"venue":null,"work_id":"6ba89fcb-33d2-4eef-85be-251c909f9130","year":2025},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.324835Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:74ad4e820760d421a25fcbb0ceb65ced539761f8c6266e971a6412e16004b151","observation_id":"7018b7cd-c429-475c-9e22-f23b80a847c8","resolution":{"observed_at":"2026-08-15T17:56:00.928020Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.327858Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.327858Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:d3521cae03f026ef31932c77bbdef729e744bf95315040b4aea4b29419f69fd0","observation_id":"570e8b7e-9123-421f-b1c5-e1fd350b8cc8","resolution":{"observed_at":"2026-08-15T17:55:59.327858Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.03233","last_updated":"2026-01-06T18:24:41Z","snapshot_observed_at":"2026-08-18T20:29:58.857701Z","submitted_at":"2026-01-06T18:24:41Z","title":"LTX-2: Efficient Joint Audio-Visual Foundation Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.03233","snapshot_observed_at":"2026-08-15T17:55:59.331009Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.331009Z"},"links":{"cited_paper":"/paper/2601.03233","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:72c3fc5558a075847da056e4c5fce952e3510acd77df130b94ef4fb3e2f1bbc4","observation_id":"f3c3e246-4ef3-43ac-bb92-34ae5321e3b5","resolution":{"observed_at":"2026-08-15T17:55:59.331009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.908475Z","title":"M.; Carrington, P.; Zimmermann, R.; and Chen, J","venue":null,"work_id":"7a733961-645d-4779-b7e6-af7c5824ee03","year":2026},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.334048Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:b1030ea20b3e846afb1481a39a572a9623aca487640cb076dea4e00c5fe547c7","observation_id":"e96f71d4-8667-46ff-8d41-10dc01ae0e7a","resolution":{"observed_at":"2026-08-15T17:56:00.911751Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.11709","last_updated":"2026-03-09T06:44:42Z","snapshot_observed_at":"2026-08-12T03:35:51.024147Z","submitted_at":"2025-05-16T21:34:47Z","title":"EgoDex: Learning Dexterous Manipulation from Large-Scale Egocentric Video","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.11709","snapshot_observed_at":"2026-08-15T17:55:59.337608Z","title":"J.; Sivapurapu, M.; and Zhang, J","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.337608Z"},"links":{"cited_paper":"/paper/2505.11709","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:8a1ef3e35f03bbf57960e4e885b7ac4f74eed39211094db3cf3e9d44aa9e4f89","observation_id":"da083a24-34d3-4460-967d-55504f0e78f5","resolution":{"observed_at":"2026-08-15T17:55:59.337608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.340600Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.340600Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:d98818f0e32f1d66d69db0894309830125a58ee236d7413b0bd3852c74266295","observation_id":"dcc081cf-f0e1-42e5-91e2-37172e9b1530","resolution":{"observed_at":"2026-08-15T17:55:59.340600Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.892667Z","title":null,"venue":null,"work_id":"1d53b73f-2dfc-4203-818d-be6b24dbf9ef","year":2024},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":101,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.343721Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:044458351446f95381a003a982d268ebf44657683ebfc4def8bcfd419ba8320b","observation_id":"b632bb93-48f9-4298-abfb-12dc922f6b97","resolution":{"observed_at":"2026-08-15T17:56:00.896517Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.882275Z","title":null,"venue":null,"work_id":"daa9e8fc-ac08-4242-9222-e7117b87e9d4","year":2026},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":102,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.346849Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:14a4675accd157c082124969b627c43f7a80b914963384f92711d6b23933bf34","observation_id":"16845c75-6a34-4c8b-bd06-05110f4d6977","resolution":{"observed_at":"2026-08-15T17:56:00.885934Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.350574Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":103,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.350574Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:005f6673b2cb06cd35bb5d9e9fd43bf2a3d79a694d87ebd8c5e64ab8ae81d4fc","observation_id":"cab18f85-eb47-4a8e-b58b-cac8d7ed5f9c","resolution":{"observed_at":"2026-08-15T17:55:59.350574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03603","last_updated":"2025-03-11T08:14:25Z","snapshot_observed_at":"2026-08-17T07:44:10.213698Z","submitted_at":"2024-12-03T23:52:37Z","title":"HunyuanVideo: A Systematic Framework For Large Video Generative Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.03603","snapshot_observed_at":"2026-08-15T17:55:59.353623Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":104,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.353623Z"},"links":{"cited_paper":"/paper/2412.03603","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:4d56678e9a2f74d38393406d8bb95f3bff534c3ad1a70cca1601b9cf0bc80b0f","observation_id":"31cfc55a-ba47-4bc2-98d1-ec230d073701","resolution":{"observed_at":"2026-08-15T17:55:59.353623Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.866894Z","title":null,"venue":null,"work_id":"a4975e79-683e-4785-b4b5-c62f21e78645","year":2022},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":105,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.356634Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:428d9376bb951a6c6d7c6509e024757e7c225759f193a0b77c0959f63300eae9","observation_id":"d06d35f5-97ed-4a38-b50b-f97d9cbea203","resolution":{"observed_at":"2026-08-15T17:56:00.870030Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.09976","last_updated":"2025-08-13T17:43:34Z","snapshot_observed_at":"2026-08-16T10:24:50.531684Z","submitted_at":"2025-08-13T17:43:34Z","title":"Masquerade: Learning from In-the-wild Human Videos using Data-Editing","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.09976","snapshot_observed_at":"2026-08-15T17:55:59.359646Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":106,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.359646Z"},"links":{"cited_paper":"/paper/2508.09976","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:1db9554cfaf55b442252809d48947a9cfc7be4ebc76007727a8b04831e9ed1c1","observation_id":"5527bd61-7d73-4c2a-a897-5668fbf66328","resolution":{"observed_at":"2026-08-15T17:55:59.359646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.857705Z","title":null,"venue":null,"work_id":"d35a55c5-d73a-4c86-9905-da1812466570","year":2026},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":107,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.362675Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:66bd21dcc9fd882c80f83f8ad7140779d33ca2f94413c02d6df67265d9f05f87","observation_id":"68cd5df0-aeaf-44d7-9669-3f86b8ff84ff","resolution":{"observed_at":"2026-08-15T17:56:00.861153Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.365768Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":108,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.365768Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:20e21164d0dad94711b69439d5c0c591e94066380f6254ea90554dab7232d99b","observation_id":"e1ea2bca-930d-4fd3-867a-92e1f531fdbc","resolution":{"observed_at":"2026-08-15T17:55:59.365768Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.370803Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":109,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.370803Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:78564f6fa2565747d5252dbfd9a761baffce94874c0f68a010b0f0f23167cb6d","observation_id":"c42722bd-7e52-4e3e-b46f-7811d1eee4ff","resolution":{"observed_at":"2026-08-15T17:55:59.370803Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.01600","last_updated":"2026-06-01T02:56:09Z","snapshot_observed_at":"2026-07-06T23:42:07.508474Z","submitted_at":"2026-06-01T02:56:09Z","title":"RoboTrustBench: Benchmarking the Trustworthiness of Video World Models for Robotic Manipulation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.01600","snapshot_observed_at":"2026-08-15T17:55:59.373923Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":110,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.373923Z"},"links":{"cited_paper":"/paper/2606.01600","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:8118da65ed0e7a63e7d4edeee543a321e15fbdffa395072cb3ec0493db967735","observation_id":"c792df27-b42d-4617-8d78-726a7c3297e0","resolution":{"observed_at":"2026-08-15T17:55:59.373923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.377263Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":111,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.377263Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:01429150e4175f38b38812aa370e65e1d9014d1db0f3517ab34ac2b79a476ed7","observation_id":"93f50e0b-c04b-4e2a-8fa2-8893a87b5bab","resolution":{"observed_at":"2026-08-15T17:55:59.377263Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.380451Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":112,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.380451Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:4fa532328ad84f8f8b4fd39dd1cec45f6535e8f295799f5e215b5eebdcf701b8","observation_id":"9dfe0813-1558-40f4-87e8-e70e0d171868","resolution":{"observed_at":"2026-08-15T17:55:59.380451Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.846950Z","title":null,"venue":null,"work_id":"803f15a1-5f3e-4ee2-9786-578a26d45253","year":2023},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":113,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.383769Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:fe99356ffad82dabbb146e8464b7a158a1ce108304afa42fa4baa5dfef89bf96","observation_id":"3a3e2199-9612-4895-ab78-f4eff82e7d8f","resolution":{"observed_at":"2026-08-15T17:56:00.850675Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.835736Z","title":null,"venue":null,"work_id":"d9656ac3-61b8-4ef4-aad5-06f3e572ec07","year":2024},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":114,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.387051Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:c35486e9acabc9b5c289e468039ba74c495d778efe762200dc5ca40d5b225a86","observation_id":"0a3469d4-8db5-48e0-b707-6a1676e73dea","resolution":{"observed_at":"2026-08-15T17:56:00.839612Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2606.11683","last_updated":"2026-06-10T05:52:14Z","snapshot_observed_at":"2026-08-17T03:39:46.869507Z","submitted_at":"2026-06-10T05:52:14Z","title":"Reason, Then Re-reason: Cross-view Revisiting Improves Spatial Reasoning","version":1},"cited_work":{"arxiv_id":"2606.11683","doi":null,"metadata_source":"pith","pith_arxiv_id":"2606.11683","snapshot_observed_at":"2026-08-15T17:56:00.578015Z","title":"Reason, Then Re-reason: Cross-view Revisiting Improves Spatial Reasoning","venue":"cs.CV","work_id":"f48b21b7-0f50-44c0-b4da-4ff2cd529359","year":2026},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":115,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.390329Z"},"links":{"cited_paper":"/paper/2606.11683","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:f4831af7dbbdb0e4f15f423c08d5dc615f77fa5f75182d3924a536ef08b9e35e","observation_id":"e93fcd93-7905-47db-b728-b29cdd1d5447","resolution":{"observed_at":"2026-08-15T17:56:00.581919Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.825748Z","title":null,"venue":null,"work_id":"01a34cb4-0fc9-456f-9c4c-73ca37a6132e","year":2023},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":116,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.393828Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:5ad59297f9bb2c8eb6de1e91dda448d57a1947e138b8522d6fff938d903cc3bf","observation_id":"a94f98de-d5d5-470b-ad4d-727ecdb2ab24","resolution":{"observed_at":"2026-08-15T17:56:00.829374Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.815464Z","title":null,"venue":null,"work_id":"4031dfa1-1a1f-4ed4-8618-4d021ce29182","year":2022},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":117,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.397154Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:a03923cae4aa1f9bea3925be7f8d7d1468fd2794b204c2853181f56429f06fc8","observation_id":"424e2829-2ed9-4de5-a218-1af8bf5d1b6e","resolution":{"observed_at":"2026-08-15T17:56:00.819123Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.27621","last_updated":"2026-04-30T09:11:25Z","snapshot_observed_at":"2026-08-16T13:03:46.974839Z","submitted_at":"2026-04-30T09:11:25Z","title":"Robot Learning from Human Videos: A Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.27621","snapshot_observed_at":"2026-08-15T17:55:59.400257Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":118,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.400257Z"},"links":{"cited_paper":"/paper/2604.27621","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:e2b44002ab8d00e92b4b37e2cc3bf9f099643b9e2d035ff4ff53865b03aa9183","observation_id":"dafff0a1-d206-499b-a24c-ac5a832ab7ef","resolution":{"observed_at":"2026-08-15T17:55:59.400257Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.804912Z","title":"J.; Kumar, V.; Zhang, A.; Bastani, O.; and Jayaraman, D","venue":null,"work_id":"c6da2fcb-713b-4c1f-9ab7-6fb826838af6","year":2023},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":119,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.403686Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:8e2a0be2825cf5163e4b5e5bfcabd0959e1eafe691e6b1f4c41f97dc7f14df3d","observation_id":"b83c57f2-3173-4dff-a464-fdb518edfe4f","resolution":{"observed_at":"2026-08-15T17:56:00.808632Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.00030","last_updated":"2023-03-07T02:29:59Z","snapshot_observed_at":"2026-08-11T15:42:16.204049Z","submitted_at":"2022-09-30T18:14:07Z","title":"VIP: Towards Universal Visual Reward and Representation via Value-Implicit Pre-Training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.00030","snapshot_observed_at":"2026-08-15T17:55:59.406973Z","title":"J.; Sodhani, S.; Jayaraman, D.; Bastani, O.; Kumar, V.; and Zhang, A","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":120,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.406973Z"},"links":{"cited_paper":"/paper/2210.00030","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:f6f20e7b5925ceb89c90ec92c7d91bfa94bf85ac1562295fdd2c0f6aa72ba1b1","observation_id":"56caf4a6-ae9e-4adc-a87b-13f0eb109a02","resolution":{"observed_at":"2026-08-15T17:55:59.406973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.794781Z","title":null,"venue":null,"work_id":"43d933c5-6ec1-45f7-b7f4-6f0fe776c147","year":2025},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":121,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.410003Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:9f77d9370d92141ab7c4878d0a692d64f4f11b3b0b37954aa6e2b02460edb719","observation_id":"9280155d-ddc7-4dcb-9111-2e3af32d4e58","resolution":{"observed_at":"2026-08-15T17:56:00.798216Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.12601","last_updated":"2022-11-18T05:57:09Z","snapshot_observed_at":"2026-08-11T08:36:53.636356Z","submitted_at":"2022-03-23T17:55:09Z","title":"R3M: A Universal Visual Representation for Robot Manipulation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.12601","snapshot_observed_at":"2026-08-15T17:55:59.413254Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":122,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.413254Z"},"links":{"cited_paper":"/paper/2203.12601","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:5ecab681e75ca83853a7100c02ab27b243f65c121d8c6ddf5517023f2601d9d5","observation_id":"ddf71642-db24-4274-afaa-52fc4f068778","resolution":{"observed_at":"2026-08-15T17:55:59.413254Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.416261Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":123,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.416261Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:11212f97be6f2bfdd2f0e991c881f4849f013b90ac9f2fdda23d1a91b915c9df","observation_id":"a6acef89-ff8d-493b-ba49-989cd7296224","resolution":{"observed_at":"2026-08-15T17:55:59.416261Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.419433Z","title":"W.; Hallacy, C.; Ramesh, A.; Goh, G.; Agarwal, S.; Sastry, G.; Askell, A.; Mishkin, P.; Clark, J.; et al","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":124,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.419433Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:9355db5e5fd05e0609048973c00da30f66e13e1e0b5edc4c0e8e17263034e405","observation_id":"51ea18c1-9a40-4620-a1ee-64e22eb0b793","resolution":{"observed_at":"2026-08-15T17:55:59.419433Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.14148","last_updated":"2026-04-15T17:59:40Z","snapshot_observed_at":"2026-08-19T04:53:25.142828Z","submitted_at":"2026-04-15T17:59:40Z","title":"Seedance 2.0: Advancing Video Generation for World Complexity","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.14148","snapshot_observed_at":"2026-08-15T17:55:59.422545Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":125,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.422545Z"},"links":{"cited_paper":"/paper/2604.14148","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:6d6757bfc2fc5505fd9635b3d7c2f90fe012401d07e8a04f778b47899367bae3","observation_id":"48dbc5c6-bdaf-410c-9072-213ea3f932e8","resolution":{"observed_at":"2026-08-15T17:55:59.422545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.425393Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":126,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.425393Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:b0e7bd527b5300b1d2860c1da3f3053d1e501ee0d60af47466b07dedf16123cc","observation_id":"78de87c9-13a0-4700-ac11-be785d160d86","resolution":{"observed_at":"2026-08-15T17:55:59.425393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.11810","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.642205Z","title":null,"venue":null,"work_id":"897824e5-448b-46d9-88a9-15b2ba93c305","year":2026},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":127,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.428935Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:519ddccd0bb05891160d68cdb891107b5bb507644502a18929f8f93516adcc85","observation_id":"7eafcf32-ea50-45a3-8a71-1da4c0ebcb41","resolution":{"observed_at":"2026-08-15T17:55:59.647745Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1912.04443","last_updated":"2020-06-21T20:16:28Z","snapshot_observed_at":"2026-08-19T08:11:43.434400Z","submitted_at":"2019-12-10T01:36:18Z","title":"AVID: Learning Multi-Stage Tasks via Pixel-Level Translation of Human Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1912.04443","snapshot_observed_at":"2026-08-15T17:55:59.431785Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":128,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.431785Z"},"links":{"cited_paper":"/paper/1912.04443","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:07971cef693137c8ce3524d7bf145ffeb8c2615110d30ee175c311bf6542e15b","observation_id":"09a8d377-090d-47b4-84e0-eccfd87b6ec0","resolution":{"observed_at":"2026-08-15T17:55:59.431785Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.434729Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":129,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.434729Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:93605eb5a57dec561cb890af35c6601bac22d07c1722a30b833063713a32ab2c","observation_id":"2f6ab971-3dd8-4731-9de8-cf6c16470a6d","resolution":{"observed_at":"2026-08-15T17:55:59.434729Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:56:00.766576Z","title":null,"venue":null,"work_id":"3ba18a48-0708-4810-8686-4f46d2217956","year":2025},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":130,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.438051Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:e536e093df8fca62d22483eaf5fcd1509979ffeb8c6100421497fee46c32274f","observation_id":"bed659b4-5921-4b5a-8e33-ac7b1b5d6aae","resolution":{"observed_at":"2026-08-15T17:56:00.769884Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.16776","last_updated":"2025-12-18T17:08:12Z","snapshot_observed_at":"2026-08-18T19:56:03.713377Z","submitted_at":"2025-12-18T17:08:12Z","title":"Kling-Omni Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2512.16776","snapshot_observed_at":"2026-08-15T17:55:59.442343Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":131,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.442343Z"},"links":{"cited_paper":"/paper/2512.16776","citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:9be6d569697a8ca57ac96bb18209fb963aff398360507f9f39228ddf8ca2ec13","observation_id":"709672ad-ec81-48d8-ab11-8c6695618917","resolution":{"observed_at":"2026-08-15T17:55:59.442343Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T17:55:59.446190Z","title":"L.; Cai, X.; Huang, Q.; Kang, Z.; Li, H.; Liang, S.; Ma, L.; Ren, S.; Wei, X.; Xie, R.; et al","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models","version":1},"reference_index":132,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:59.446190Z"},"links":{"citing_paper":"/paper/2608.13049"},"observation_digest":"sha256:8a540f660caa6833a8470d0bd97c2e5eec0d6329a05c4fbc8305d2ee9620fda6","observation_id":"4d5218e4-6363-49bf-b0c5-3a74b5d985b8","resolution":{"observed_at":"2026-08-15T17:55:59.446190Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2608.13049","last_updated":"2026-08-13T10:14:33Z","latest_version":1,"primary_category":"cs.RO","snapshot_observed_at":"2026-08-19T08:11:53.286748Z","submitted_at":"2026-08-13T10:14:33Z","title":"H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":1,"unresolved":87,"verified_exact":2,"verified_fuzzy":10},"total_outbound_references":117},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 100 of 117 outbound references and 0 inbound Pith citation observations for arXiv:2608.13049."}