{"as_of":"2026-08-08T01:11:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0e3638f589263e9aed041eaa7aed08d987400ccb06f4382e9d347ef272e62455","coverage":[{"denominator":150,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T23:53:47.129799Z","state":"measured"},{"denominator":102,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":102,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T16:55:22.682553Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-01T16:15:49.358084Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.04681","snapshot_observed_at":"2026-08-03T16:55:22.682553Z","title":"Perceiving and acting in first-person: A dataset and benchmark for egocentric human-object-human interac- tions.arXiv preprint arXiv:2508.04681, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.11393","last_updated":"2026-06-10T09:16:33Z","snapshot_observed_at":"2026-08-07T01:29:09.387766Z","submitted_at":"2025-12-12T09:07:21Z","title":"The N-Body Problem: Parallel Execution from Single-Person Egocentric Video","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-03T16:55:22.682553Z"},"links":{"cited_paper":"/paper/2508.04681","citing_paper":"/paper/2512.11393"},"observation_digest":"sha256:0ec668ca300c41c64029f7c8f7c526e03bb8bfe043e80330974b9b32b960fe31","observation_id":"d714f7d3-0497-4b75-a585-5fd7ff7201f3","resolution":{"observed_at":"2026-08-03T16:55:22.682553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"cited_work":{"arxiv_id":"2508.04681","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.04681","snapshot_observed_at":"2026-07-01T16:15:49.358084Z","title":"Perceiving and acting in first-person: A dataset and benchmark for ego- centric human-object-human interactions.arXiv preprint arXiv:2508.04681, 2025","venue":null,"work_id":"b2532e91-1bb1-4d14-87f4-11e45e4a4117","year":2025},"citing_paper":{"arxiv_id":"2606.28604","last_updated":"2026-06-26T20:52:54Z","snapshot_observed_at":"2026-08-06T06:50:02.421167Z","submitted_at":"2026-06-26T20:52:54Z","title":"IMU-HOI: A Symbiotic Framework for Coherent Human-Object Interaction and Motion Capture via Contact-Conscious Inertial Fusion","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-30T00:50:08.637598Z"},"links":{"cited_paper":"/paper/2508.04681","citing_paper":"/paper/2606.28604"},"observation_digest":"sha256:4f04a27b3c27a26f1220cfdc2a99150a638d6378639a25563f5350cc5021b233","observation_id":"d1c5ad6a-301d-4350-87c4-3972e9f07fb9","resolution":{"observed_at":"2026-07-01T16:15:49.359573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2508.04681/citation-record","integrity":"/paper/2508.04681/integrity","json":"/paper/2508.04681/citation-record.json","paper":"/paper/2508.04681"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:36.377873Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:36.377873Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:5f04c58a3347f859a1235ea6191f34b11a04b9fc665e7f39c05774a64a05286b","observation_id":"aa1de90f-0463-4f3c-987c-a75187a329bd","resolution":{"observed_at":"2026-08-05T23:53:36.377873Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:36.493641Z","title":"https://eth-ait.github.io/aitviewer/","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:36.493641Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:df35d5b0b4c84779b99b9420dab00204f46223e151d36c42afef406837bfcc89","observation_id":"c2267eaa-8907-4e8b-a8f9-b145b8535e9d","resolution":{"observed_at":"2026-08-05T23:53:36.493641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:36.580669Z","title":"https://github.com/zju3dv/EasyMocap","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:36.580669Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:b20270b2fe6aceb53b26575286fa3f2eca34a8775d9d2184b191d911e565a2e6","observation_id":"637fe4ef-1ad5-416f-835d-91cc9aee4fdd","resolution":{"observed_at":"2026-08-05T23:53:36.580669Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:36.644157Z","title":"https://github.com/opencv/opencv/blob/master/data/haarcascades/haarcascade_frontalface_default.xml","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:36.644157Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:a2d1a56fba70dc316d38ff10ae15668fd3eaa9b66908d959d36b0f7db3236a91","observation_id":"53604ed2-fd65-4f0c-b0d0-dfec27dc6b28","resolution":{"observed_at":"2026-08-05T23:53:36.644157Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:36.677317Z","title":"3d human pose perception from egocentric stereo videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:36.677317Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:f5d337cb1aea8601885c4f9333163939ac19c66cad1fa95bc7931914f26e548b","observation_id":"3afd4ab0-2388-4769-a490-b969f4fce32f","resolution":{"observed_at":"2026-08-05T23:53:36.677317Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:36.791573Z","title":"Rgbmanip: Monocular image-based robotic manipulation through active object pose estimation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:36.791573Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:843a7968fc8abe2631365a93994e81c35858ff4911cf94b19c63bde6e1dac74a","observation_id":"7c0eb52a-6ee3-477f-a726-e254bf10104a","resolution":{"observed_at":"2026-08-05T23:53:36.791573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:36.912396Z","title":"Circle: Capture in rich contextual environments","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:36.912396Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:9b00e89457144696efe32eb5ce7846d349fbf3cf45b39e6579da67f3f02ac603","observation_id":"df6e3a0a-faba-4978-b8c6-1b3dd8a1e718","resolution":{"observed_at":"2026-08-05T23:53:36.912396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09598","last_updated":"2024-06-13T21:38:17Z","snapshot_observed_at":"2026-07-06T18:30:41.802045Z","submitted_at":"2024-06-13T21:38:17Z","title":"Introducing HOT3D: An Egocentric Dataset for 3D Hand and Object Tracking","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.09598","snapshot_observed_at":"2026-08-05T23:53:36.990477Z","title":"Introducing hot3d: An egocentric dataset for 3d hand and object tracking","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:36.990477Z"},"links":{"cited_paper":"/paper/2406.09598","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:b42601b430ac52f45b477a25b18557e3b2cbd68e33cbd0ff87f267bd8165fcf5","observation_id":"7f94ed07-29b9-48c9-9422-497b4960be4b","resolution":{"observed_at":"2026-08-05T23:53:36.990477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:37.078642Z","title":"Uncertainty-aware state space transformer for egocentric 3d hand trajectory forecasting","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.078642Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:6c137aa8c1dbc50149d07f87b8dda27a81dc260ba7d9906d51934f5b4cb37358","observation_id":"c3be9f17-9051-40ea-98b3-2b3c435f46f6","resolution":{"observed_at":"2026-08-05T23:53:37.078642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:37.182153Z","title":"Gen2act: Human video generation in novel scenarios enables generalizable robot manipulation, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.182153Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:42f9815b23661122c37db66b5eadbd08cd2aa24f8c245af644ea07f98c5ca89f","observation_id":"7199a25e-d890-4589-ad57-fea792773787","resolution":{"observed_at":"2026-08-05T23:53:37.182153Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:37.277795Z","title":"Behave: Dataset and method for tracking human object interactions","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.277795Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:07dd3999d1142a2d4765e8c8203840a70e1046b0d23cb53833eb55f4220a5d7b","observation_id":"741097dd-6161-418c-a120-9e1924c9d250","resolution":{"observed_at":"2026-08-05T23:53:37.277795Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:37.330804Z","title":"Bedlam: A synthetic dataset of bodies exhibiting detailed lifelike animated motion","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.330804Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:8b5b88a48d424812e9100dee7ae09d4cc36191763bb328a184127d695c94e804","observation_id":"fc5b2d54-9aa7-446d-8358-ee8b86c909dc","resolution":{"observed_at":"2026-08-05T23:53:37.330804Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:37.408257Z","title":"Contactpose: A dataset of grasps with object contact and hand pose","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.408257Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:73d984150c5e3e2d27a4d21a9b92ec5bdc31c6b6c06dd3e50b860829f69e412a","observation_id":"3c868352-74a1-441c-9d95-d5f2e52f6a37","resolution":{"observed_at":"2026-08-05T23:53:37.408257Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.06817","last_updated":"2023-08-11T17:45:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-12-13T18:55:15Z","title":"RT-1: Robotics Transformer for Real-World Control at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.06817","snapshot_observed_at":"2026-08-05T23:53:37.497994Z","title":"Rt-1: Robotics transformer for real-world control at scale","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.497994Z"},"links":{"cited_paper":"/paper/2212.06817","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:3beaccf0da8d2a3cf17e8dddd5be240f88f62c20251b26281be46d85c200251c","observation_id":"23ea0b4e-5697-4c39-b686-95bcc92198fe","resolution":{"observed_at":"2026-08-05T23:53:37.497994Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15818","last_updated":"2023-07-28T21:18:02Z","snapshot_observed_at":"2026-08-02T16:17:50.621617Z","submitted_at":"2023-07-28T21:18:02Z","title":"RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15818","snapshot_observed_at":"2026-08-05T23:53:37.605325Z","title":"Rt-2: Vision-language-action models transfer web knowledge to robotic control","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.605325Z"},"links":{"cited_paper":"/paper/2307.15818","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:089007f964b88ab105f4d04540b33202e51f1fc73169c4ce2abe34ede9ee5aff","observation_id":"63beb163-f63b-4281-81d6-8cbe59e62bb1","resolution":{"observed_at":"2026-08-05T23:53:37.605325Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:37.748118Z","title":"u tepage, Ali Ghadirzadeh, \\","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.748118Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:8daca3b880cc8f3b60414cde95b87ba9d8886237abd141618fefaf9317fdaae7","observation_id":"c0f2382a-e4e0-427a-a6fc-21c90bf34ea7","resolution":{"observed_at":"2026-08-05T23:53:37.748118Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:37.811974Z","title":"Understanding hand-object manipulation with grasp types and object attributes","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.811974Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:a7cdd05504b50ea4281b66d3cb433150d695e24def484362be1e135b629ab739","observation_id":"b738690b-82b7-4577-aa62-669705241df2","resolution":{"observed_at":"2026-08-05T23:53:37.811974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:37.924543Z","title":"Long-term human motion prediction with scene context","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:37.924543Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:0f651bf8cecee4dbb99f0dd2a67f80d6f8cce8b2b8593d43e9c90b04ba42b566","observation_id":"67efd82f-7bf1-43b7-aec3-34a8ef337e8a","resolution":{"observed_at":"2026-08-05T23:53:37.924543Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:38.044463Z","title":"A multi-sensor dataset of human-human handover","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:38.044463Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:4923c4f0b45ab2782cd87732f060f556f8dfaa7fec6c068de3faa8625f7df58a","observation_id":"4528d23f-6faf-4075-b840-84a1f362cc51","resolution":{"observed_at":"2026-08-05T23:53:38.044463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:38.135981Z","title":"On the choice of grasp type and location when handing over an object","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:38.135981Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:0a9063c47127540079016022c86778758008c116ea89edd658d3471e71a26213","observation_id":"0dd07cb0-67c7-4e6a-a318-1aff0c879f18","resolution":{"observed_at":"2026-08-05T23:53:38.135981Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:38.221352Z","title":"Context-aware human motion prediction, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:38.221352Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:86af9ac2ce26c2999e4e71a635435963361e5afe24d990154143510246e0b92f","observation_id":"b97bbdaa-52e9-4d73-9327-38f0ffcff497","resolution":{"observed_at":"2026-08-05T23:53:38.221352Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:38.359155Z","title":"Towards collaborative robots as intelligent co-workers in human-robot joint tasks: what to do and who does it? In ISR 2020; 52th International Symposium on Robotics, pages 1--8","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:38.359155Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:6e13602a0736b72519e4867b1756daa478e224e3a015a04eea2ddf03e18113d7","observation_id":"acd28e02-93ff-42f6-9bb9-009146e79861","resolution":{"observed_at":"2026-08-05T23:53:38.359155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:38.499590Z","title":"You-do, i-learn: Egocentric unsupervised discovery of objects and their modes of interaction towards video-based guidance","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:38.499590Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:ae39c2178fb92f40fabe6e67e7a03c3c59068e30d4ace005aa3e2dd5a9481d01","observation_id":"4e870b8a-891a-4f0c-8d30-7d70ab1c02c1","resolution":{"observed_at":"2026-08-05T23:53:38.499590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:38.632786Z","title":"Scaling egocentric vision: The epic-kitchens dataset","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:38.632786Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:ae4ae3e9a016f0459aafe4de297ad2a6c265dd6f0c0e3699795912c6983ec125","observation_id":"8aa3846d-1915-4897-9826-96ed3c690a18","resolution":{"observed_at":"2026-08-05T23:53:38.632786Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:38.729105Z","title":"Rescaling egocentric vision: Collection, pipeline and challenges for epic-kitchens-100","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:38.729105Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:d213c72b0773943be0f82e5bd5f8944259f02ed5c803031f1e2f0461e7307057","observation_id":"2b403280-8c96-4083-aab9-05cadc26cf86","resolution":{"observed_at":"2026-08-05T23:53:38.729105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16097","last_updated":"2024-05-17T15:00:55Z","snapshot_observed_at":"2026-07-06T16:53:18.275859Z","submitted_at":"2023-11-27T18:59:10Z","title":"CG-HOI: Contact-Guided 3D Human-Object Interaction Generation","version":2},"cited_work":{"arxiv_id":"2311.16097","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.16097","snapshot_observed_at":"2026-08-05T23:53:56.465466Z","title":"CG-HOI: Contact-Guided 3D Human-Object Interaction Generation","venue":"cs.CV","work_id":"dd3c5a9f-03d9-4f06-9d4c-45908ff4d9f0","year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:38.839860Z"},"links":{"cited_paper":"/paper/2311.16097","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:baca9a135f190668b7adb2f95f765dc826982f210d6b8c7b9d53faff6224db7d","observation_id":"7cc1fa15-4d4e-400c-aaab-c35637204c28","resolution":{"observed_at":"2026-08-05T23:53:56.623103Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.03378","last_updated":"2023-03-06T18:58:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-06T18:58:06Z","title":"PaLM-E: An Embodied Multimodal Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.03378","snapshot_observed_at":"2026-08-05T23:53:38.955903Z","title":"Palm-e: An embodied multimodal language model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:38.955903Z"},"links":{"cited_paper":"/paper/2303.03378","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:ff25de2b24cfd914ac7d361245761a429cef9fc1969319dbfac831cb47a232dd","observation_id":"443523c8-f5f6-4272-96b1-ba82e3f4812c","resolution":{"observed_at":"2026-08-05T23:53:38.955903Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:39.065926Z","title":"Avatars grow legs: Generating smooth human motion from sparse tracking inputs with diffusion model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:39.065926Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:7ef57ff4724a31e51562f94382fc4c33152828e8567e16548fd3e23e3ec54759","observation_id":"dcf469ee-718a-41ce-b577-8169e806610d","resolution":{"observed_at":"2026-08-05T23:53:39.065926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:39.194833Z","title":"Human preferences for robot eye gaze in human-to-robot handovers","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:39.194833Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:a77aceed0a838e2cfb719cf1f70ddaaf6315696af17997d0d96ce30b648b5131","observation_id":"3621407e-7517-41e5-af35-72f97fc86ba3","resolution":{"observed_at":"2026-08-05T23:53:39.194833Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.07095","last_updated":"2025-07-09T17:52:04Z","snapshot_observed_at":"2026-08-06T18:45:14.871573Z","submitted_at":"2025-07-09T17:52:04Z","title":"Go to Zero: Towards Zero-shot Motion Generation with Million-scale Data","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.07095","snapshot_observed_at":"2026-08-05T23:53:39.305639Z","title":"Go to zero: Towards zero-shot motion generation with million-scale data","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:39.305639Z"},"links":{"cited_paper":"/paper/2507.07095","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:0783db1433196cf14bde3524eb5763a372500cffa86dce1a7e7d5898233dc312","observation_id":"096698dc-b191-4aaf-9133-c0261f37b320","resolution":{"observed_at":"2026-08-05T23:53:39.305639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:39.444849Z","title":"Arctic: A dataset for dexterous bimanual hand-object manipulation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:39.444849Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:9245942cd9af1903229dbfdad8792125e0c5f66ac5fe6f3aa228eda2ed4155eb","observation_id":"07301309-3c82-42da-b6e6-9914b5ac4b6a","resolution":{"observed_at":"2026-08-05T23:53:39.444849Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:39.587309Z","title":"Hold: Category-agnostic 3d reconstruction of interacting hands and objects from video","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:39.587309Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:a44d8b2c79ba76d110527a193a6d3d3d33743ae1fffdf0cbb8fe3be875039d39","observation_id":"dee6ca9c-d63a-40b4-9876-41f032fbe3a6","resolution":{"observed_at":"2026-08-05T23:53:39.587309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:39.790354Z","title":"Benchmarks and challenges in pose estimation for egocentric hand interactions with objects","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:39.790354Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:25099e1da395bb505ab98dfa64442411e2e438e53ef45b122c119b61bf4105aa","observation_id":"7ae5b3cf-4111-4bbf-a0fe-16e696c5106b","resolution":{"observed_at":"2026-08-05T23:53:39.790354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:39.864859Z","title":"Social interactions: A first-person perspective","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:39.864859Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:ca36d1bd11106a8dbea7591c548cd87270ad05948a46217fc7e960e472615916","observation_id":"6335ad8c-60e1-43dd-8fbb-2aa0a2c81dcf","resolution":{"observed_at":"2026-08-05T23:53:39.864859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:39.991773Z","title":"Three-dimensional reconstruction of human interactions","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:39.991773Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:7d9f5c74e85cdc4d8185a42db4fa0ba55fe03715b2fbd88e46dabe88e463c708","observation_id":"0eb12c38-8be0-483c-b62a-d2a8f1ca5486","resolution":{"observed_at":"2026-08-05T23:53:39.991773Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:40.138119Z","title":"Ego4d: Around the world in 3,000 hours of egocentric video","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:40.138119Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:01f8e9905864b3a964021ba26bfb9c84fcd907900326b731ed69ed298d77c954","observation_id":"e122517d-adc9-4f43-a2fb-ee0c6eb16bb6","resolution":{"observed_at":"2026-08-05T23:53:40.138119Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:40.326420Z","title":"Ego-exo4d: Understanding skilled human activity from first-and third-person perspectives","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:40.326420Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:0e1aa5f342ec3f23bfa26b22b4afc66533e949b705d219b9feb26861d53e776a","observation_id":"214b7387-8c1f-40c9-a230-a1df3f6076c2","resolution":{"observed_at":"2026-08-05T23:53:40.326420Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:40.453469Z","title":"Generating diverse and natural 3d human motions from text","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:40.453469Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:5222f096636998d215ca39ae6c2d1d2f335c7bb518856700726c275198b5fcca","observation_id":"c801ee8b-b65b-42f0-9bb3-28c76597692a","resolution":{"observed_at":"2026-08-05T23:53:40.453469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:40.557001Z","title":"Multi-person extreme motion prediction","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:40.557001Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:4296d83798215beddb11ae51fa43c7551eea685b85fb888eab6b03e4bc515ba0","observation_id":"e7441a50-ee9e-476d-ba80-d6d19eb6cf4c","resolution":{"observed_at":"2026-08-05T23:53:40.557001Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:40.686731Z","title":"Interaction replica: Tracking human--object interaction and scene changes from human motion","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:40.686731Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:e21b28eacc8cfdb908058ab9d3c85749d9472dcac7687f5b2c42093da83d5d8f","observation_id":"a9369fba-bce0-4115-8971-d5d5a7a3f543","resolution":{"observed_at":"2026-08-05T23:53:40.686731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:40.802399Z","title":"Honnotate: A method for 3d annotation of hand and object poses","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:40.802399Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:fec27c2a25c5f6741093da51450f1147f9b4691f13b21e7c30bc473e255376e6","observation_id":"86d875fc-d8c8-4c6d-9cbb-3e8d77a093fc","resolution":{"observed_at":"2026-08-05T23:53:40.802399Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:40.944657Z","title":"Keypoint transformer: Solving joint identification in challenging hands and object interactions for accurate 3d pose estimation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:40.944657Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:c12012653cddb707fb5e9fecc1e069e1e77294af7db2f996bfcaea7ba90ffb72","observation_id":"fdf4964c-ff11-4cba-8b00-09b2d2f91658","resolution":{"observed_at":"2026-08-05T23:53:40.944657Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.08530","last_updated":"2024-10-11T05:02:31Z","snapshot_observed_at":"2026-07-06T19:31:36.072099Z","submitted_at":"2024-10-11T05:02:31Z","title":"Ego3DT: Tracking Every 3D Object in Ego-centric Videos","version":1},"cited_work":{"arxiv_id":"2410.08530","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.08530","snapshot_observed_at":"2026-08-05T23:53:56.111211Z","title":"Ego3DT: Tracking Every 3D Object in Ego-centric Videos","venue":"cs.CV","work_id":"2a6cce85-0162-4af3-92fd-d67d43e94f6e","year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:41.136974Z"},"links":{"cited_paper":"/paper/2410.08530","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:5f4fa72f73d8e75d580e50e7e015a6030d2fef285922a7a07dc9800960f7178a","observation_id":"0ff40bc8-b9fe-42e7-adba-4c6dc91f5fa9","resolution":{"observed_at":"2026-08-05T23:53:56.245318Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:41.282187Z","title":"Resolving 3d human pose ambiguities with 3d scene constraints","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:41.282187Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:e5c61464c16dbca1f60a9ec418a9bab7a7ee7e5314304f3494186968dc0bff9e","observation_id":"e5f1267f-bf0e-405a-8a15-8108ec6c2095","resolution":{"observed_at":"2026-08-05T23:53:41.282187Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:41.447218Z","title":"Stochastic scene-aware motion prediction","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:41.447218Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:df28e667f5539a6fc9249c8020f7dd27b2ac19cd47cc5dc3f4d83b83b9e8c1b1","observation_id":"15d62b75-0a72-4770-a012-7f12306552ba","resolution":{"observed_at":"2026-08-05T23:53:41.447218Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08858","last_updated":"2024-06-13T06:44:46Z","snapshot_observed_at":"2026-08-06T19:58:25.663767Z","submitted_at":"2024-06-13T06:44:46Z","title":"OmniH2O: Universal and Dexterous Human-to-Humanoid Whole-Body Teleoperation and Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08858","snapshot_observed_at":"2026-08-05T23:53:41.596866Z","title":"Omnih2o: Universal and dexterous human-to-humanoid whole-body teleoperation and learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:41.596866Z"},"links":{"cited_paper":"/paper/2406.08858","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:0e8a4e2389f93941f2e13e64fd606d94840094788d8011f8d5e3115ca7f6a609","observation_id":"0582ff3c-d9bc-45c1-a624-24c9cc183149","resolution":{"observed_at":"2026-08-05T23:53:41.596866Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:41.794282Z","title":"Learning human-to-humanoid real-time whole-body teleoperation, 2024 b","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:41.794282Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:4863db7f6c79975d4dc00f480412e428d088b30feac685408a95d8ab5999fb74","observation_id":"4b32be66-0bfe-4137-ac4b-c8c75fbb0f73","resolution":{"observed_at":"2026-08-05T23:53:41.794282Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:42.007046Z","title":"Gans trained by a two time-scale update rule converge to a local nash equilibrium","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:42.007046Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:83e8d2cac12e2c42fef43c5211bf65edc449aba24a563d61ed70ea1a36029117","observation_id":"2856fd24-9f52-4d4c-9065-a998b32df01a","resolution":{"observed_at":"2026-08-05T23:53:42.007046Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.05973","last_updated":"2023-11-02T06:53:37Z","snapshot_observed_at":"2026-08-05T01:03:23.456778Z","submitted_at":"2023-07-12T07:40:48Z","title":"VoxPoser: Composable 3D Value Maps for Robotic Manipulation with Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.05973","snapshot_observed_at":"2026-08-05T23:53:42.151513Z","title":"Voxposer: Composable 3d value maps for robotic manipulation with language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:42.151513Z"},"links":{"cited_paper":"/paper/2307.05973","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:de62974c944afa83e5c81f1586e571558eeb81d20fa183f5fa3642988efeed9b","observation_id":"4311e1c5-9ae5-4490-b670-d5993dfb874a","resolution":{"observed_at":"2026-08-05T23:53:42.151513Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:42.262006Z","title":"Intercap: Joint markerless 3d tracking of humans and objects in interaction","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:42.262006Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:d3ab2598b39d4dab66414ea01a2dc18c36372326e8370609d6560b4624598ae2","observation_id":"f19ee7a2-a298-47de-920a-c60f261c0c49","resolution":{"observed_at":"2026-08-05T23:53:42.262006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:42.305300Z","title":"Human3.6m: Large scale datasets and predictive methods for 3d human sensing in natural environments","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:42.305300Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:d6b6e2cef9fb9fce3e82c01e4405ed8f2a26ef3f7dd31897ec8fc69e17513d4a","observation_id":"e45659d0-a564-4095-95ac-982b45f4d7ed","resolution":{"observed_at":"2026-08-05T23:53:42.305300Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:42.493342Z","title":"A large-scale rgb-d database for arbitrary-view human action recognition","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:42.493342Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:097fa9f5267092decb31e1016be568aa3c8a252eea4c9ee6739121eb44145a9e","observation_id":"540747ee-8105-4d24-a3c2-00fd13e7c80c","resolution":{"observed_at":"2026-08-05T23:53:42.493342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:42.668418Z","title":"Affordpose: A large-scale dataset of hand-object interactions with affordance-driven hand pose","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:42.668418Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:ad441ee24b2afafce241d9e5932685838a628351c3ea52780a5ea7888157c294","observation_id":"4a821793-7172-4353-8844-4f6f6468a9ca","resolution":{"observed_at":"2026-08-05T23:53:42.668418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:42.751703Z","title":"Avatarposer: Articulated full-body pose tracking from sparse motion sensing","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:42.751703Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:f2fad2c8a541e3c3670fa3d2e071562f681f6f1a17bc27f856045d172ecfefd7","observation_id":"c29d00bc-a3cc-4a5b-ba66-8aedf8ef1dd0","resolution":{"observed_at":"2026-08-05T23:53:42.751703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06493","last_updated":"2024-09-06T11:28:04Z","snapshot_observed_at":"2026-07-06T16:05:31.757410Z","submitted_at":"2023-08-12T07:46:50Z","title":"EgoPoser: Robust Real-Time Egocentric Pose Estimation from Sparse and Intermittent Observations Everywhere","version":3},"cited_work":{"arxiv_id":"2308.06493","doi":null,"metadata_source":"pith","pith_arxiv_id":"2308.06493","snapshot_observed_at":"2026-08-05T23:53:55.751984Z","title":"EgoPoser: Robust Real-Time Egocentric Pose Estimation from Sparse and Intermittent Observations Everywhere","venue":"cs.CV","work_id":"96b98d4b-cbd6-485e-aff2-dd06d1130a48","year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:42.826886Z"},"links":{"cited_paper":"/paper/2308.06493","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:6ee0ed406c63842bd08f7f87d73be74c24c693d148f646cc113155f18c73e4d7","observation_id":"02026fc3-0c18-47a5-ba10-343bd0d8cf6c","resolution":{"observed_at":"2026-08-05T23:53:55.933190Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:42.954752Z","title":"Full-body articulated human-object interaction","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:42.954752Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:0ab3ebab086eeeacb33392e3636063e55f3a4e9a9c70a70b3004e51068f6e322","observation_id":"db12ccd5-e4f1-4f79-84d4-a32bc36b14ab","resolution":{"observed_at":"2026-08-05T23:53:42.954752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:43.115113Z","title":"Scaling up dynamic human-scene interaction modeling","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.115113Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:6a57805533df8f632d84fc21d4d7309df80428a428c0caeec5e2e1a35ebbd971","observation_id":"04876618-26f0-4a7f-8572-fd8d25a1593f","resolution":{"observed_at":"2026-08-05T23:53:43.115113Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:43.225202Z","title":"A probabilistic attention model with occlusion-aware texture regression for 3d hand reconstruction from a single rgb image","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.225202Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:a0f194cd2b8c8935237d0bf3701667938def19bf92baf7d15c052b31c1e9cc94","observation_id":"934c652c-9301-4a15-83ee-91d1737144ce","resolution":{"observed_at":"2026-08-05T23:53:43.225202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.12773","last_updated":"2024-10-16T17:48:50Z","snapshot_observed_at":"2026-08-07T23:49:15.433379Z","submitted_at":"2024-10-16T17:48:50Z","title":"Harmon: Whole-Body Motion Generation of Humanoid Robots from Language Descriptions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.12773","snapshot_observed_at":"2026-08-05T23:53:43.249143Z","title":"Harmon: Whole-body motion generation of humanoid robots from language descriptions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.249143Z"},"links":{"cited_paper":"/paper/2410.12773","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:30f86ae9745b4be7d245ff0ac07835d28cf8d8a96a8df0da2dc5f3cff1df977b","observation_id":"272642f2-06d5-4d6c-bd35-960ce43b776d","resolution":{"observed_at":"2026-08-05T23:53:43.249143Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:43.317534Z","title":"Epic-fusion: Audio-visual temporal binding for egocentric action recognition","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.317534Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:c410cab99acb7d0e0bc9f686a02a81a6b02420ff00b9b74e5ebafdb9259769bb","observation_id":"221c6dde-b73b-41db-a401-66a69b69baf1","resolution":{"observed_at":"2026-08-05T23:53:43.317534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2011.13851","last_updated":"2020-11-27T17:29:48Z","snapshot_observed_at":"2026-07-06T10:18:25.252546Z","submitted_at":"2020-11-27T17:29:48Z","title":"Real-time Active Vision for a Humanoid Soccer Robot Using Deep Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"2011.13851","doi":null,"metadata_source":"pith","pith_arxiv_id":"2011.13851","snapshot_observed_at":"2026-08-05T23:53:55.456916Z","title":"Real-time Active Vision for a Humanoid Soccer Robot Using Deep Reinforcement Learning","venue":"cs.RO","work_id":"1a884a38-b66d-487a-a572-b0f901193c2b","year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.434526Z"},"links":{"cited_paper":"/paper/2011.13851","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:8f19e295f72233898242d187a060d7a4fa3d8ed09a4ca7e5285988bffbecbdbb","observation_id":"5d01fd9d-6d53-4e3d-bb7c-5a4f07b5556f","resolution":{"observed_at":"2026-08-05T23:53:55.571899Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09246","last_updated":"2024-09-05T19:46:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-13T15:46:55Z","title":"OpenVLA: An Open-Source Vision-Language-Action Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.09246","snapshot_observed_at":"2026-08-05T23:53:43.518630Z","title":"Openvla: An open-source vision-language-action model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.518630Z"},"links":{"cited_paper":"/paper/2406.09246","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:7366b878706cc79fe73ce0ee20c58d425e11f8e82559a03fb0f0058b8882088d","observation_id":"d38ec4df-1d52-4a30-bbdf-9b42b17424f1","resolution":{"observed_at":"2026-08-05T23:53:43.518630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:43.577785Z","title":"Dataset of bimanual human-to-human object handovers","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.577785Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:dba90888c45c43b44b87beaf7924593bafad1fc3f749a8ddbf51edc541dd3d9c","observation_id":"4bad86b4-1f7a-4b57-91c4-76c86cf57f97","resolution":{"observed_at":"2026-08-05T23:53:43.577785Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:43.620869Z","title":"H2o: Two hands manipulating objects for first person interaction recognition","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.620869Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:e8122067a1d6e265bf2b3d355bf8a73abf0f24245a77989a95f57de793d3be59","observation_id":"74d306ed-8824-4528-8da0-04a41b9ab49a","resolution":{"observed_at":"2026-08-05T23:53:43.620869Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:43.657151Z","title":"Ego-body pose estimation via ego-head pose estimation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.657151Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:73a16fa2309095d9514ce74c5f529b699bc2106106eabae6dd8d560a2feefec1","observation_id":"1634a6f9-e40d-459d-8a68-83d1b510bbac","resolution":{"observed_at":"2026-08-05T23:53:43.657151Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:43.710565Z","title":"Dngaussian: Optimizing sparse-view 3d gaussian radiance fields with global-local depth normalization","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.710565Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:fd943a7417816ed48671a65f1ef81fed653784c567eb31e415df874fd7a25166","observation_id":"6ac6dd36-1ea9-4af2-94f3-aed9b7e4c3ea","resolution":{"observed_at":"2026-08-05T23:53:43.710565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:43.836230Z","title":"In the eye of beholder: Joint learning of gaze and actions in first person video","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.836230Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:c32c49ade93fbae693ec6baa20a4671280799e141aceee4d2d536af2d67a2e16","observation_id":"a2cd7b7f-d1c1-4a31-8ee7-66062f9d93ef","resolution":{"observed_at":"2026-08-05T23:53:43.836230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:43.872711Z","title":"Ego-exo: Transferring visual representations from third-person to first-person videos","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:43.872711Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:ce97dc99ac8c16919d9e9864f72d68ace6a3ff435e71591e72f2e6b5415a0af3","observation_id":"85dd708c-dd21-4f9e-b8b9-e4a1c9408867","resolution":{"observed_at":"2026-08-05T23:53:43.872711Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05684","last_updated":"2024-03-28T03:15:57Z","snapshot_observed_at":"2026-07-06T15:14:44.015806Z","submitted_at":"2023-04-12T08:12:29Z","title":"InterGen: Diffusion-based Multi-human Motion Generation under Complex Interactions","version":3},"cited_work":{"arxiv_id":"2304.05684","doi":null,"metadata_source":"pith","pith_arxiv_id":"2304.05684","snapshot_observed_at":"2026-08-05T23:53:55.224988Z","title":"InterGen: Diffusion-based Multi-human Motion Generation under Complex Interactions","venue":"cs.CV","work_id":"57f7c78c-8069-470d-9122-c2febdbd4766","year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.001874Z"},"links":{"cited_paper":"/paper/2304.05684","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:57ad9aad7048a97052fe70be2355827dc326a17fed048845dfe1333c4f4246d7","observation_id":"ed506fd7-27ef-4549-bd59-405a3d77fc9e","resolution":{"observed_at":"2026-08-05T23:53:55.311121Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.00818","last_updated":"2024-01-26T15:40:29Z","snapshot_observed_at":"2026-08-03T09:56:41.811697Z","submitted_at":"2023-07-03T07:57:29Z","title":"Motion-X: A Large-scale 3D Expressive Whole-body Human Motion Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.00818","snapshot_observed_at":"2026-08-05T23:53:44.127161Z","title":"Motion-x: A large-scale 3d expressive whole-body human motion dataset","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.127161Z"},"links":{"cited_paper":"/paper/2307.00818","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:1b3e6416966d079296c4c8680d78b1d2de5f7ceee14c220f9ac352e943315f9f","observation_id":"1303fdd2-dda5-47b3-ba6b-42691f636be3","resolution":{"observed_at":"2026-08-05T23:53:44.127161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:44.251374Z","title":"Gaussian-flow: 4d reconstruction with dynamic 3d gaussian particle","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.251374Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:eb2fcfaef0631e85c0302bfe1fca6d076c0a753a24672a4748c5eadbbaa56bd2","observation_id":"ef18c462-ed8c-4f1f-b2e2-19f809f7258a","resolution":{"observed_at":"2026-08-05T23:53:44.251374Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:44.356535Z","title":"Ntu rgb+ d 120: A large-scale benchmark for 3d human activity understanding","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.356535Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:3b54fd745331e9ec551bded311ae929b6cc75c70f778b6e87dee2758ee2ab97f","observation_id":"f54d7c30-e9b5-454c-ae4b-de9ae1f6a908","resolution":{"observed_at":"2026-08-05T23:53:44.356535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:44.444765Z","title":"Forecasting human-object interaction: joint prediction of motor attention and actions in first person video","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.444765Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:a5af36ffc779e6f684fd589da9effef2baa63fdd71c9aa961df7d0e4d70d5269","observation_id":"fd84285f-2b75-4da0-9fbf-dbea7c596e92","resolution":{"observed_at":"2026-08-05T23:53:44.444765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:44.519068Z","title":"Joint hand motion and interaction hotspots prediction from egocentric videos","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.519068Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:accb25a95ee5427e3b59df0b3234f94cc7b60f775fce140232cf6ba29508779d","observation_id":"5c933f72-f545-4062-8e53-7df5675bf231","resolution":{"observed_at":"2026-08-05T23:53:44.519068Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:44.539091Z","title":"Hoi4d: A 4d egocentric dataset for category-level human-object interaction","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.539091Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:c428b774cba72cb1bf7c8e9c7071f8b7dfdb1d79bbd1f465e9a5db81a2c12699","observation_id":"3ca0b4e4-52a1-4498-8e77-195cdf414d08","resolution":{"observed_at":"2026-08-05T23:53:44.539091Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:44.630172Z","title":"Taco: Benchmarking generalizable bimanual tool-action-object understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.630172Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:3c6676f92377fa39e42ee5bcecc1309b32e8d905241b02bd3501681f3f33e633","observation_id":"f4b8d642-93a6-492e-acf9-d1e4b928a82f","resolution":{"observed_at":"2026-08-05T23:53:44.630172Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:44.744966Z","title":null,"venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.744966Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:09906f9d29b54b6f1d850dfc03135e95dce7d9e32091c1449e8da3447217d255","observation_id":"3781f457-10ca-4a54-be07-f57c3090e0ef","resolution":{"observed_at":"2026-08-05T23:53:44.744966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:44.817529Z","title":"Dynamics-regulated kinematic policy for egocentric pose estimation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.817529Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:d4fcd171a8d2ad296fada27ce2d7d52fb945271b1a7dfba43fc0f3ac1198c0ca","observation_id":"040ac900-4abe-475f-85a7-3216bd152279","resolution":{"observed_at":"2026-08-05T23:53:44.817529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:44.945085Z","title":"Himo: A new benchmark for full-body human interacting with multiple objects","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:44.945085Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:98b732a85f8406030307fd13db7ffbc686d5fc7c72d6076f54f081ce709c817f","observation_id":"02652ef5-4919-4481-b2ee-5d9781c0527b","resolution":{"observed_at":"2026-08-05T23:53:44.945085Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:45.081764Z","title":"Diff-ip2d: Diffusion-based hand-object interaction prediction on egocentric videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.081764Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:f0a84febed83dc000051a7fb3d3b7cdcb764230d0e4103f0785c653c079b3038","observation_id":"cf2c0312-2946-4b8f-864e-5076d3ef0c92","resolution":{"observed_at":"2026-08-05T23:53:45.081764Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14093","last_updated":"2026-05-01T01:50:44Z","snapshot_observed_at":"2026-08-04T06:47:25.827167Z","submitted_at":"2024-05-23T01:43:54Z","title":"A Survey on Vision-Language-Action Models for Embodied AI","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.14093","snapshot_observed_at":"2026-08-05T23:53:45.175736Z","title":"A survey on vision-language-action models for embodied ai","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.175736Z"},"links":{"cited_paper":"/paper/2405.14093","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:20781e98efd31895c9ffc70eee92050d7e0a62497019c22bc7c99c8e6f4eb680","observation_id":"0a503427-8a8a-415e-8d53-5bb314c41e3d","resolution":{"observed_at":"2026-08-05T23:53:45.175736Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:45.259308Z","title":"Unifying representations and large-scale whole-body motion databases for studying human motion","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.259308Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:31e09bb61177ecb4de184973fd66bfa7f3cf0499baf95ae3e5c6ec44e222165c","observation_id":"a7a7c949-41e9-4dce-8267-c6515b83c5f9","resolution":{"observed_at":"2026-08-05T23:53:45.259308Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:45.369610Z","title":"Dexvip: Learning dexterous grasping with human hand pose priors from video","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.369610Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:3e22c6ab0466569640390a9bf191c61e6a20e381e4d4de6dfe939b2b89cf00a1","observation_id":"59975bdf-f0ad-41d6-a98b-55029605412f","resolution":{"observed_at":"2026-08-05T23:53:45.369610Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:45.453729Z","title":"Hoi4abot: Human-object interaction anticipation for human intention reading collaborative robots, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.453729Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:ee88e587095541887b08c95289bbf56ead475c08bf3835745c9d52a78f2233b6","observation_id":"2aae58d6-a389-4889-8e72-a893b81060c9","resolution":{"observed_at":"2026-08-05T23:53:45.453729Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:45.573456Z","title":"Eventego3d: 3d human motion capture from egocentric event streams","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.573456Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:43830d67a876f3fb087ae9e230844f136c6394755966f531ed75f1812ba96a7c","observation_id":"84938642-17db-41c8-a20c-ea177fbaf462","resolution":{"observed_at":"2026-08-05T23:53:45.573456Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:45.657160Z","title":"imapper: interaction-guided scene mapping from monocular videos","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.657160Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:780884d4b4cc0912738c7a26fa922a80d7b9108cf45766565c46b7d5fc40fc01","observation_id":"494b04af-b36d-4159-9380-b432773a1f3f","resolution":{"observed_at":"2026-08-05T23:53:45.657160Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:53:45.749385Z","title":"Grounded human-object interaction hotspots from video","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.749385Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:8658300e7ff23d87b9dc70b4378d778209ff70974d1c6c935650b6e8fbae37b6","observation_id":"0ddf64cf-2b30-4952-bdf7-f01f72aaadca","resolution":{"observed_at":"2026-08-05T23:53:45.749385Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:07.722317Z","title":"Jointly learning energy expenditures and activities using egocentric multimodal signals","venue":null,"work_id":"947d016d-6ed6-4b9c-b7da-22eb8c5465a6","year":2017},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.857569Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:c685927d35c02d78bff8fcebf1c5a1f15763a32ff1a4e925f3116b24bee7435f","observation_id":"69936d59-83a1-47c4-a8df-2f5a62524333","resolution":{"observed_at":"2026-08-05T23:54:07.798225Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:07.582075Z","title":"You2me: Inferring body pose in egocentric video via first and second person interactions","venue":null,"work_id":"8e9fece5-c74f-42a3-923f-01122b091fa8","year":2020},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:45.929362Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:c35ac00da1ffc26fedd1648d4eb6ece8dbe53351f708696d3a8b3575090368f1","observation_id":"6150a92d-f71f-493e-93d4-937619f86f4e","resolution":{"observed_at":"2026-08-05T23:54:07.639114Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08864","last_updated":"2025-05-14T15:22:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-13T05:20:40Z","title":"Open X-Embodiment: Robotic Learning Datasets and RT-X Models","version":9},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08864","snapshot_observed_at":"2026-08-05T23:53:46.042835Z","title":"Open x-embodiment: Robotic learning datasets and rt-x models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.042835Z"},"links":{"cited_paper":"/paper/2310.08864","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:adbe0e01db5df82ab4ffa801c61937b6b5c8964b1e82e7eff32b7abf0696d674","observation_id":"33ef0a73-01ba-4479-920e-8b3bdf9b2981","resolution":{"observed_at":"2026-08-05T23:53:46.042835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:07.400131Z","title":"GPT -3.5 turbo fine-tuning and api updates","venue":null,"work_id":"bbd2ba3b-211f-407d-8f2d-bdf340430b06","year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.142375Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:3b28155e5b25f5b6d95f01d7119c68f9697308babd107219c7710afe031f195c","observation_id":"1afc4540-fead-4485-a4e8-854b5a4c8869","resolution":{"observed_at":"2026-08-05T23:54:07.487788Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:07.184857Z","title":"Handoccnet: Occlusion-robust 3d hand mesh estimation network","venue":null,"work_id":"90a1e57f-79c9-4454-8c96-24297042dafa","year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.246943Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:8d7df05fb12e0f66b2dba1f38f6ef615707b9e9b02b0135c1f781ae609b1e6da","observation_id":"4c35c632-a8b2-4dc0-a419-66eeda1914cd","resolution":{"observed_at":"2026-08-05T23:54:07.289841Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:06.966379Z","title":"Reconstructing hands in 3 D with transformers","venue":null,"work_id":"6704ba4a-3547-48aa-97d0-d4f763bbf6c4","year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.352960Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:ca3e1e55f5f2dcf0bca1318131519a2e4226478587919418bb609e029a57f135","observation_id":"d041e185-0ec9-4d58-aa3c-b08493b00189","resolution":{"observed_at":"2026-08-05T23:54:07.075067Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06553","last_updated":"2025-07-07T05:09:32Z","snapshot_observed_at":"2026-08-06T14:47:15.286683Z","submitted_at":"2023-12-11T17:41:17Z","title":"HOI-Diff: Text-Driven Synthesis of 3D Human-Object Interactions using Diffusion Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.06553","snapshot_observed_at":"2026-08-05T23:53:46.456426Z","title":"Hoi-diff: Text-driven synthesis of 3d human-object interactions using diffusion models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.456426Z"},"links":{"cited_paper":"/paper/2312.06553","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:c3cb252d2e6be8d767c3705f31bea72d141b6ef648431cd1916ecf12289c3d00","observation_id":"8d0e308e-1883-4061-a35a-d07bf788daf1","resolution":{"observed_at":"2026-08-05T23:53:46.456426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:06.778148Z","title":"Action-conditioned 3d human motion synthesis with transformer vae","venue":null,"work_id":"d8dc1039-d934-444b-be59-b625cd4e1917","year":2021},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.571809Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:2fe14bb90d79d2fa08b768af15cdfb84c07f38255a98c6852c91bf47149353da","observation_id":"d63f23e3-813a-48bf-aa21-5460f407a4e5","resolution":{"observed_at":"2026-08-05T23:54:06.852047Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:06.532077Z","title":"The kit motion-language dataset","venue":null,"work_id":"92d000ff-7386-444e-aeb1-e3605e93c828","year":2016},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.687039Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:9ea24e7e4e8b17833aa35b60183edd03f03c4e79f788bb5639799314b566cbf6","observation_id":"123dced8-f65b-427f-8cd9-68a3f50f93da","resolution":{"observed_at":"2026-08-05T23:54:06.629054Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12259","last_updated":"2025-03-26T18:05:52Z","snapshot_observed_at":"2026-07-06T19:17:46.809272Z","submitted_at":"2024-09-18T18:46:51Z","title":"WiLoR: End-to-end 3D Hand Localization and Reconstruction in-the-wild","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12259","snapshot_observed_at":"2026-08-05T23:53:46.756838Z","title":"Wilor: End-to-end 3d hand localization and reconstruction in-the-wild","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.756838Z"},"links":{"cited_paper":"/paper/2409.12259","citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:028529881d494e626cae80ac0c06fb10352e68366ee97484417fbc3ed16be573","observation_id":"56b393d0-dc53-4bec-9a5f-648d367cb9a3","resolution":{"observed_at":"2026-08-05T23:53:46.756838Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:06.304769Z","title":"Mild: multimodal interactive latent dynamics for learning human-robot interaction","venue":null,"work_id":"81c64d38-24cf-4f8a-9add-1012da9136b5","year":2022},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.877540Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:e157a81927a019deefeade2d6f604b63f75d24849f4b55fc0d1f2bd2a5e0a3f0","observation_id":"5415fe11-7538-45d6-ac2a-24de01eb5990","resolution":{"observed_at":"2026-08-05T23:54:06.447546Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:06.106318Z","title":"Moveint: Mixture of variational experts for learning human-robot interactions from demonstrations","venue":null,"work_id":"90458d95-6c1d-4f5e-a356-dbf4f16a78ed","year":2024},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:46.982127Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:68bdaf1cc35ff04ad94864a5b4770ced3a9a8e28c9cf247fb8dd14909a087e89","observation_id":"5241ac67-f43e-44d9-b168-caecac513f19","resolution":{"observed_at":"2026-08-05T23:54:06.202591Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:54:05.900642Z","title":"The virtual caliper: Rapid creation of metrically accurate avatars from 3d measurements","venue":null,"work_id":"9c1c183d-22b6-4d5e-9ad3-5e0d185a7384","year":2019},"citing_paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions","version":1},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-08-05T23:53:47.129799Z"},"links":{"citing_paper":"/paper/2508.04681"},"observation_digest":"sha256:94eea756b11bfbcbd0dc8d95d07d708c0e520fb14d471cbc16f7852a1bdb3c98","observation_id":"da8967d9-cbf4-48d8-af9a-09d86ffdd0a7","resolution":{"observed_at":"2026-08-05T23:54:06.000165Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2508.04681","last_updated":"2025-08-06T17:46:23Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-06T06:49:33.366476Z","submitted_at":"2025-08-06T17:46:23Z","title":"Perceiving and Acting in First-Person: A Dataset and Benchmark for Egocentric Human-Object-Human Interactions"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":85,"verified_exact":5,"verified_fuzzy":10},"total_outbound_references":150},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 100 of 150 outbound references and 2 inbound Pith citation observations for arXiv:2508.04681."}