{"as_of":"2026-08-20T09:45:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:098dd9df576040bcd071e1cfce1883c332f3e5b126fdd25fe33857363018b588","coverage":[{"denominator":42,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":42,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-28T22:04:09.296855Z","state":"measured"},{"denominator":43,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":43,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T01:34:31.334140Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.31234","snapshot_observed_at":"2026-08-02T01:34:31.334140Z","title":"HARP-VLA: Human-robot aligned representation learn- ing for vision-language-action model,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.14635","last_updated":"2026-07-16T06:59:39Z","snapshot_observed_at":"2026-08-16T00:43:16.013283Z","submitted_at":"2026-07-16T06:59:39Z","title":"Action QFormer: Structured Representation Shaping under Action Supervision in Vision-Language-Action Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-02T01:34:31.334140Z"},"links":{"cited_paper":"/paper/2605.31234","citing_paper":"/paper/2607.14635"},"observation_digest":"sha256:b42e63546a9438458a64357497f77766c9bf95843d2118c996ac331115f2694d","observation_id":"074e1e22-2d11-4cd4-b40b-49822b0b9199","resolution":{"observed_at":"2026-08-02T01:34:31.334140Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2605.31234/citation-record","integrity":"/paper/2605.31234/integrity","json":"/paper/2605.31234/citation-record.json","paper":"/paper/2605.31234"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2410.11758","last_updated":"2025-05-15T12:13:37Z","snapshot_observed_at":"2026-08-12T08:06:42.236760Z","submitted_at":"2024-10-15T16:28:09Z","title":"Latent Action Pretraining from Videos","version":2},"cited_work":{"arxiv_id":"2410.11758","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.11758","snapshot_observed_at":"2026-07-08T02:04:26.214499Z","title":"Latent Action Pretraining from Videos","venue":"cs.RO","work_id":"f32313d6-51a8-4863-8646-8e1870c15e2f","year":2024},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2410.11758","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:8699ce907c127d30a2aebbf5a09c089850c8c3b460dcc69fb53a07e2ad0689f2","observation_id":"35ba7f4b-76ac-43b1-8817-eabe773a0b30","resolution":{"observed_at":"2026-07-01T19:46:10.862338Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.06111","last_updated":"2025-11-03T11:52:57Z","snapshot_observed_at":"2026-08-09T00:51:47.897197Z","submitted_at":"2025-05-09T15:11:13Z","title":"UniVLA: Learning to Act Anywhere with Task-centric Latent Actions","version":3},"cited_work":{"arxiv_id":"2505.06111","doi":"10.48550/arxiv.2505.06111","metadata_source":"pith","pith_arxiv_id":"2505.06111","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"UniVLA: Learning to Act Anywhere with Task-centric Latent Actions","venue":"cs.RO","work_id":"e05d654d-db73-48f6-9318-381b6798bac9","year":2025},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2505.06111","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:9169015f3bc24e3b74977a8d311e105db2e14b02abde40b5aa44ab7ca50d9549","observation_id":"32aa5940-6bd1-4895-baeb-e3d3a16511a7","resolution":{"observed_at":"2026-07-01T19:46:10.880387Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.08787","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T06:49:38.585919Z","title":"J., and Lee, Y","venue":null,"work_id":"92a2a59c-da64-4564-a9b1-e2d2e4a2ca65","year":2025},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:05aae2c5c84897366a69c189ced535e3472bf193d345af92a5014c4d78208809","observation_id":"7798bbd5-02ca-4d48-a471-b698704aeccd","resolution":{"observed_at":"2026-07-01T19:46:10.857254Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T22:04:09.296855Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:a79999a02468ad3d18d1aa14c9f5efe6dc025f52a7f8c13e56bea6a0dfcee43b","observation_id":"43bdcb58-c2db-4c92-8681-e5411df3c993","resolution":{"observed_at":"2026-06-28T22:04:09.296855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T22:04:09.296855Z","title":"Kareer, D","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:eb47bdf1c7123074ad2543860e191a10bf1d3087bd266d2bfd3b42e935255480","observation_id":"064d0ce0-74bc-4ff7-8b8e-abfaca8ea110","resolution":{"observed_at":"2026-06-28T22:04:09.296855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.22199","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T23:37:43.468061Z","title":"arXiv preprint arXiv:2509.22199 (2025)","venue":null,"work_id":"57481d5b-7951-4649-8091-ad5137278a74","year":2025},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:e8895c60abc8fbd3ae1b58d6cbe9e48c77111b4f44177fd8602ebb4630994e3c","observation_id":"fd2b0dbb-915d-462e-a9a2-8558caa50aa8","resolution":{"observed_at":"2026-07-01T19:46:10.860090Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.21864","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T06:04:34.623476Z","title":"Dexumi: Using human hand as the universal manipulation in- terface for dexterous manipulation","venue":null,"work_id":"4d12dc64-2d49-4e60-828d-2647f059c29b","year":2025},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:8024192c27a24e57f2947baa1274bf7c64b7207cce844c3a80c19c5807843c5f","observation_id":"6a093d09-7fb5-4210-a04e-d5fdfc8e3699","resolution":{"observed_at":"2026-07-01T19:46:10.864905Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T22:04:09.296855Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:456a8f776f323cac1a9bfd7448afd51e246d6a5ddd7a85958ce0d1534c17b24d","observation_id":"42fc5a09-bc12-4092-b6b8-cc41e2272440","resolution":{"observed_at":"2026-06-28T22:04:09.296855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09246","last_updated":"2024-09-05T19:46:34Z","snapshot_observed_at":"2026-08-16T21:53:14.144225Z","submitted_at":"2024-06-13T15:46:55Z","title":"OpenVLA: An Open-Source Vision-Language-Action Model","version":3},"cited_work":{"arxiv_id":"2406.09246","doi":"10.18653/v1/2022.naacl-main.68","metadata_source":"pith","pith_arxiv_id":"2406.09246","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OpenVLA: An Open-Source Vision-Language-Action Model","venue":"cs.RO","work_id":"3e7e65c5-5aed-4fe9-8414-2092bcb31cc7","year":2024},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2406.09246","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:657ef51e33d42be61f52264c73525c2ed060076c0f080f3f9494da4efc87d9ee","observation_id":"d194bc2e-f932-440a-a3fe-6bbec9073162","resolution":{"observed_at":"2026-07-01T19:46:10.864485Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.06817","last_updated":"2023-08-11T17:45:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-12-13T18:55:15Z","title":"RT-1: Robotics Transformer for Real-World Control at Scale","version":2},"cited_work":{"arxiv_id":"2212.06817","doi":"10.48550/arxiv.2212.06817","metadata_source":"pith","pith_arxiv_id":"2212.06817","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RT-1: Robotics Transformer for Real-World Control at Scale","venue":"cs.RO","work_id":"e11bda85-8531-46bc-a07f-d0ade3643ab1","year":2022},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2212.06817","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:bc71a9d7e59ffb50933aaab892971259a102c81d7c3f879082bc38476ab9a4ea","observation_id":"a03984bb-8b93-466d-b361-620c6c29afe8","resolution":{"observed_at":"2026-07-01T19:46:10.876832Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15818","last_updated":"2023-07-28T21:18:02Z","snapshot_observed_at":"2026-08-02T16:17:50.621617Z","submitted_at":"2023-07-28T21:18:02Z","title":"RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control","version":1},"cited_work":{"arxiv_id":"2307.15818","doi":"10.48550/arxiv.2307.15818","metadata_source":"pith","pith_arxiv_id":"2307.15818","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control","venue":"cs.RO","work_id":"ff438a8a-8003-4fae-9131-acd418b3597b","year":2023},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2307.15818","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:179b68a159d396ffe320045f909b38b83aec06dab57860b49edf1a96bd739b18","observation_id":"e01e983c-4254-4ae8-a537-c8858f3b617c","resolution":{"observed_at":"2026-07-01T19:46:10.861914Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T22:04:09.296855Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:5833b05facb22b6e50fa7e3453d5d5fabec533cfb66561eb97c77194f411383f","observation_id":"1c0df870-7193-4d33-bc0d-30c4d0db2665","resolution":{"observed_at":"2026-06-28T22:04:09.296855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12213","last_updated":"2024-05-26T19:55:26Z","snapshot_observed_at":"2026-08-14T08:22:11.295012Z","submitted_at":"2024-05-20T17:57:01Z","title":"Octo: An Open-Source Generalist Robot Policy","version":2},"cited_work":{"arxiv_id":"2405.12213","doi":"10.48550/arxiv.2405.12213","metadata_source":"pith","pith_arxiv_id":"2405.12213","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Octo: An Open-Source Generalist Robot Policy","venue":"cs.RO","work_id":"f9ca0722-8855-48c3-a27a-0eefb7e19253","year":2024},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2405.12213","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:a44761619a098d71189d75533e2202982bd854809a3653b894ba1c09dc88637d","observation_id":"93e23b68-3d6b-4038-a72b-1225fb19e32c","resolution":{"observed_at":"2026-07-01T19:46:10.859283Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07864","last_updated":"2025-03-01T08:57:15Z","snapshot_observed_at":"2026-08-20T09:40:44.825072Z","submitted_at":"2024-10-10T12:33:46Z","title":"RDT-1B: a Diffusion Foundation Model for Bimanual Manipulation","version":2},"cited_work":{"arxiv_id":"2410.07864","doi":"10.48550/arxiv.2410.07864","metadata_source":"pith","pith_arxiv_id":"2410.07864","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RDT-1B: a Diffusion Foundation Model for Bimanual Manipulation","venue":"cs.RO","work_id":"12319725-bc7d-4c32-a229-ad270a7460bc","year":2024},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2410.07864","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:15511cfde8ed6bab51ae03e2329fb69a87dad663f9b21c368b5199723c68bb58","observation_id":"dc4e691f-b7f0-493f-a413-67f31826560a","resolution":{"observed_at":"2026-07-01T19:46:10.878008Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.24164","last_updated":"2026-01-08T17:01:05Z","snapshot_observed_at":"2026-08-16T17:53:54.636855Z","submitted_at":"2024-10-31T17:22:30Z","title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","version":4},"cited_work":{"arxiv_id":"2410.24164","doi":"10.48550/arxiv.2410.24164","metadata_source":"pith","pith_arxiv_id":"2410.24164","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","venue":"cs.LG","work_id":"f790abdc-a796-482f-a40d-f8ee035ecfc2","year":2024},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2410.24164","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:b244178aa5de1694346bb7450e5e5d997ba501e4f9cd884fa38d2aabf048a59c","observation_id":"0ea0dfb9-827f-4003-ba3f-9db05ebc6cad","resolution":{"observed_at":"2026-07-01T19:46:10.869549Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.16054","last_updated":"2025-04-22T17:31:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-22T17:31:29Z","title":"$\\pi_{0.5}$: a Vision-Language-Action Model with Open-World Generalization","version":1},"cited_work":{"arxiv_id":"2504.16054","doi":"10.1609/aaai.v40i28.39562","metadata_source":"pith","pith_arxiv_id":"2504.16054","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"$\\pi_{0.5}$: a Vision-Language-Action Model with Open-World Generalization","venue":"cs.LG","work_id":"d1ad7304-d09a-49bc-809e-846439f6aff9","year":2025},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2504.16054","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:84f044e73382da37a33e7c9fab422d5033186aad803e0641d3ddd193d65e825b","observation_id":"5879ffed-fee5-4786-8183-47c259e5ac3f","resolution":{"observed_at":"2026-07-01T19:46:10.872735Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.01977","last_updated":"2023-11-06T05:53:08Z","snapshot_observed_at":"2026-08-16T14:46:11.532367Z","submitted_at":"2023-11-03T15:31:51Z","title":"RT-Trajectory: Robotic Task Generalization via Hindsight Trajectory Sketches","version":2},"cited_work":{"arxiv_id":"2311.01977","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.01977","snapshot_observed_at":"2026-07-04T17:09:59.550952Z","title":"Rt-trajectory: Robotic task generalization via hindsight trajectory sketches","venue":null,"work_id":"cfaa2d68-1521-4065-b84f-c32f5a2e061d","year":2023},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2311.01977","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:362bb6f149513d68f6a563fc47cb87b304e53a4c465901ede96ee8427de569d3","observation_id":"70717396-c379-4ad4-be3c-847929f6a063","resolution":{"observed_at":"2026-07-01T19:46:10.828518Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T22:04:09.296855Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:b64450af280828e179db5feee57cfe62b29e1ecf0901408af5620815a8debf30","observation_id":"36f83a56-0072-4c87-b74b-9317167cb482","resolution":{"observed_at":"2026-06-28T22:04:09.296855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T22:04:09.296855Z","title":"Xiong, Q","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:670aa7c9abdac2bdc04e5d389c3f7adee37c0940ecad17adc3fae6b82331efbe","observation_id":"66491bf7-482b-402a-8085-bb8907fa9795","resolution":{"observed_at":"2026-06-28T22:04:09.296855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15208","last_updated":"2024-10-04T04:05:27Z","snapshot_observed_at":"2026-08-16T13:32:21.178898Z","submitted_at":"2024-07-21T16:15:02Z","title":"Flow as the Cross-Domain Manipulation Interface","version":2},"cited_work":{"arxiv_id":"2407.15208","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.15208","snapshot_observed_at":"2026-07-10T18:47:31.575403Z","title":"Flow as the cross-domain manipulation interface","venue":"cs.RO","work_id":"40196361-5c47-4912-87a6-0e600288d197","year":2024},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2407.15208","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:b945b133c4828dbeb32b90d0fdda907f2e4705c93641180fdc55e4ad7fc1d8eb","observation_id":"ad161016-7bb7-4fe6-ad2b-bb91d39a2f9b","resolution":{"observed_at":"2026-07-01T19:46:10.870093Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.24766","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T18:47:31.619682Z","title":"Dream2flow: Bridging video generation and open-world manipulation with 3d object flow","venue":null,"work_id":"574eb7d1-d6e1-4e88-920b-5439943aee71","year":2025},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:c85f91dfc439d1e195e1059fac7f7a3f4bd56b6364e74d1a72fd641147184917","observation_id":"053c5aec-d184-4b12-89d2-c6a7fffd8c09","resolution":{"observed_at":"2026-07-01T19:46:10.879323Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.03403","last_updated":"2024-09-09T03:11:19Z","snapshot_observed_at":"2026-08-16T13:21:07.574304Z","submitted_at":"2024-09-05T10:39:15Z","title":"RoVi-Aug: Robot and Viewpoint Augmentation for Cross-Embodiment Robot Learning","version":2},"cited_work":{"arxiv_id":"2409.03403","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.03403","snapshot_observed_at":"2026-07-04T01:19:20.797647Z","title":"Rovi-aug: Robot and viewpoint augmentation for cross-embodiment robot learning","venue":null,"work_id":"79bb3206-d811-400c-85a6-e3f0a6aaa851","year":2024},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2409.03403","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:6a91e6401ec249c14674d573a9222d928ee301a7e4b3838844a98346814a246e","observation_id":"a5149324-bddb-46b0-8e8f-b2ba96f0aca5","resolution":{"observed_at":"2026-07-01T19:46:10.875390Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T22:04:09.296855Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:dbda4306a330e886a6b5583065f5d0c9df26d911a8ca18490a7e44cbde1b45ed","observation_id":"0c48d54d-9fdb-4815-a340-8eb285135d4a","resolution":{"observed_at":"2026-06-28T22:04:09.296855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.05513","last_updated":"2025-09-05T21:47:55Z","snapshot_observed_at":"2026-08-15T14:04:16.295023Z","submitted_at":"2025-09-05T21:47:55Z","title":"OpenEgo: A Large-Scale Multimodal Egocentric Dataset for Dexterous Manipulation","version":1},"cited_work":{"arxiv_id":"2509.05513","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.05513","snapshot_observed_at":"2026-07-04T13:19:50.405733Z","title":"Openego: A large-scale multimodal egocentric dataset for dexterous manipulation","venue":null,"work_id":"74521572-d4e8-4a70-b113-46c1d0c7906a","year":2025},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2509.05513","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:03a6e35960512d7e25da0c8b89c0b053ba1b5c595bda9141a86e78529a097286","observation_id":"6f7be1f5-8fa4-4301-92af-9add02daba39","resolution":{"observed_at":"2026-07-01T19:46:10.853926Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T22:04:09.296855Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:8984e8c235648bb667345610f22d672e46ee6d078948c5e551db191936e778bc","observation_id":"d98d2698-de49-45ff-85cd-5f29bd1ba444","resolution":{"observed_at":"2026-06-28T22:04:09.296855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.00595","last_updated":"2023-09-26T10:47:35Z","snapshot_observed_at":"2026-08-17T23:49:22.371557Z","submitted_at":"2023-07-02T15:33:31Z","title":"RH20T: A Comprehensive Robotic Dataset for Learning Diverse Skills in One-Shot","version":2},"cited_work":{"arxiv_id":"2307.00595","doi":"10.48550/arxiv.2307.00595","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.00595","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RH20T: A comprehensive robotic dataset for learning diverse skills in one-shot","venue":"arXiv (Cornell University)","work_id":"d0404087-2d28-482f-8b50-8400d35d315e","year":2023},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2307.00595","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:f813e469f245d480697c9bd33ea22b55f1ac617da4a29d54d8a24b50ef3b9604","observation_id":"dc173fc5-0de5-45e0-b9d2-7a24665d9ae0","resolution":{"observed_at":"2026-07-01T19:46:10.866879Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2502.16587","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T04:09:35.384218Z","title":"Human2robot: Learning robot actions from paired human-robot videos","venue":null,"work_id":"6366a195-66a4-4661-84a0-a9103b4d43a4","year":2025},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:4bdcdc791bd99a540f544213e534244a0d23a5c41805969673bbb39bed5ab68d","observation_id":"0528053c-90e5-4e9a-a3be-2f2609f06edc","resolution":{"observed_at":"2026-07-01T19:46:10.840796Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T22:04:09.296855Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:812901fc465f8361f19815091fa8cb59c92a5eed73dc1a333ff91c392a58ac95","observation_id":"3d1d54bc-cab0-4e2a-8457-312c3ba68754","resolution":{"observed_at":"2026-06-28T22:04:09.296855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T22:04:09.296855Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:58a48615e5ad83512174a889a105c1dc924df4105e28c1c1d2fda53895596eb8","observation_id":"bcd5b816-09a3-4b20-9c41-c3958f8bc118","resolution":{"observed_at":"2026-06-28T22:04:09.296855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T22:04:09.296855Z","title":"James, Z","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:e4c636f132ae0b83635dd0729eb13a19a3828f7de5231d5de968dac720046002","observation_id":"f5e5bddf-bb26-453c-aaaf-b6070511d466","resolution":{"observed_at":"2026-06-28T22:04:09.296855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19645","last_updated":"2025-04-28T07:49:39Z","snapshot_observed_at":"2026-08-15T09:35:08.116329Z","submitted_at":"2025-02-27T00:30:29Z","title":"Fine-Tuning Vision-Language-Action Models: Optimizing Speed and Success","version":2},"cited_work":{"arxiv_id":"2502.19645","doi":"10.48550/arxiv.2502.19645","metadata_source":"pith","pith_arxiv_id":"2502.19645","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Fine-Tuning Vision-Language-Action Models: Optimizing Speed and Success","venue":"cs.RO","work_id":"04f46bb3-4346-47e8-bf09-c75d91f96e87","year":2025},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2502.19645","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:cd3d4b1e7fcd8afc0dc7b7346e57449ead83276d6589b563c2ff27d0643a8d7f","observation_id":"b9d80d99-ee8b-42f0-8c7a-ff17d73a253a","resolution":{"observed_at":"2026-07-01T19:46:10.874408Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.20795","last_updated":"2026-05-29T12:22:36Z","snapshot_observed_at":"2026-08-13T21:51:15.131045Z","submitted_at":"2025-05-27T06:56:14Z","title":"Learning Generalizable Robot Policy with Human Demonstration Video as a Prompt","version":2},"cited_work":{"arxiv_id":"2505.20795","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.20795","snapshot_observed_at":"2026-07-01T19:46:10.870685Z","title":"Learning Generalizable Robot Policy with Human Demonstration Video as a Prompt","venue":"cs.RO","work_id":"4d2b064a-236a-4510-822b-f8b6cd04ee39","year":2025},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2505.20795","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:986db8a78356e9f3aba77cb5aa316969bcbbe0d947732362d20fd752a23c3f5d","observation_id":"2fdecf46-b87f-4a73-8443-9f2a24b1aaad","resolution":{"observed_at":"2026-07-01T19:46:10.871892Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":"2505.09388","doi":"10.1016/j.aiopen.2022.12","metadata_source":"pith","pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3 Technical Report","venue":"cs.CL","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","year":2025},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:7e3c91f2fe195e6b484153641e0ad8909a77b68d1294ae9524db560328fc368f","observation_id":"cd3b6cb3-e617-48f6-9a23-d167f338a5e0","resolution":{"observed_at":"2026-07-01T19:46:10.828901Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T22:04:09.296855Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:075a5377a6320d0f89023a0f29d15e72128a45da2b3ae7147baa193e1127c07c","observation_id":"3d7ee676-4f8c-4a4e-9f65-577cb22c9551","resolution":{"observed_at":"2026-06-28T22:04:09.296855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T22:04:09.296855Z","title":"Doersch, Y","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:02f30556dc4c013b0ea4496bec546e2f04f2f459657bc55d78bd12416b896b7d","observation_id":"13b58a9c-ff64-4a76-9aa0-11893ee77d72","resolution":{"observed_at":"2026-06-28T22:04:09.296855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T22:04:09.296855Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:fec95b0a0222029664e84397a22d6265b526a786fd1fba765b7075d57a516a7c","observation_id":"9c1cc4fc-6f01-4c0e-9c69-abcbfed54641","resolution":{"observed_at":"2026-06-28T22:04:09.296855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2201.02610","last_updated":"2022-01-07T18:59:32Z","snapshot_observed_at":"2026-08-16T17:29:40.380982Z","submitted_at":"2022-01-07T18:59:32Z","title":"Embodied Hands: Modeling and Capturing Hands and Bodies Together","version":1},"cited_work":{"arxiv_id":"2201.02610","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2201.02610","snapshot_observed_at":"2026-07-04T09:39:46.884388Z","title":"Em- bodied hands: Modeling and capturing hands and bodies to- gether","venue":null,"work_id":"82cb27bc-fcfa-4002-b59a-ec2d3366a168","year":2022},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2201.02610","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:6ad913580beda564531385550f90ac99e479ebe835b8791d5c028a198e945588","observation_id":"e63e87b2-d476-41d7-898f-25fd56ec8bec","resolution":{"observed_at":"2026-07-01T19:46:10.842693Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T22:04:09.296855Z","title":"Yadav and M","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:eedddde87a14b2fc9c50dda524d10beae57de9dfe53170e1758feb46046ab3fe","observation_id":"e9d5f882-8369-41c2-812e-47a56fd1ab28","resolution":{"observed_at":"2026-06-28T22:04:09.296855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T22:04:09.296855Z","title":"Karamcheti, S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:498e7b8e193c4ea261840342277e4f7e562bafd67763ff8bfdf4d867190d6560","observation_id":"aa901d56-61d9-4944-afe0-b4583f30dfaa","resolution":{"observed_at":"2026-06-28T22:04:09.296855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.07193","last_updated":"2024-02-02T10:24:09Z","snapshot_observed_at":"2026-08-17T13:03:40.359628Z","submitted_at":"2023-04-14T15:12:19Z","title":"DINOv2: Learning Robust Visual Features without Supervision","version":2},"cited_work":{"arxiv_id":"2304.07193","doi":"10.48550/arxiv.2304.07193","metadata_source":"pith","pith_arxiv_id":"2304.07193","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DINOv2: Learning Robust Visual Features without Supervision","venue":"cs.CV","work_id":"26b304e5-b54a-4f26-be7e-83299eca52e4","year":2023},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2304.07193","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:75d3d4e8edd38f6ef516a9a67117b33007b903e96b940c274265006b960c2f07","observation_id":"47cfa813-6f33-4c28-a66f-e58eb02e9f15","resolution":{"observed_at":"2026-07-01T19:46:10.850895Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T22:04:09.296855Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:386f8227f85ed3349f2bb6f2485411ed975b50d3d34ddc30e02866273b24e409","observation_id":"2fa1af2e-1e3d-43e8-83c9-0860a7a943a2","resolution":{"observed_at":"2026-06-28T22:04:09.296855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":"2307.09288","doi":"10.24963/ijcai.2025/706","metadata_source":"pith","pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","venue":"cs.CL","work_id":"68a5177f-d644-44c1-bd4f-4e5278c22f5d","year":2023},"citing_paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-28T22:04:09.296855Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2605.31234"},"observation_digest":"sha256:12341e26916ca7cea35b8f389ca91052629da8ff86af9303946536209909e349","observation_id":"4a3a4bea-dcba-45ba-a093-82660c00d795","resolution":{"observed_at":"2026-07-01T19:46:10.834458Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.31234","last_updated":"2026-05-29T12:36:30Z","latest_version":1,"primary_category":"cs.RO","snapshot_observed_at":"2026-08-16T14:26:50.597994Z","submitted_at":"2026-05-29T12:36:30Z","title":"HARP-VLA: Human-Robot Aligned Representation Learning for Vision-Language-Action Model"},"reference_resolution":{"displayed":42,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":17,"verified_exact":25,"verified_fuzzy":0},"total_outbound_references":42},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 42 of 42 outbound references and 1 inbound Pith citation observation for arXiv:2605.31234."}