{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5MLRLRK6H2CUF5QVZR3O34HL53","short_pith_number":"pith:5MLRLRK6","schema_version":"1.0","canonical_sha256":"eb1715c55e3e8542f615cc76edf0ebeefbd918649b754e7c0f96b5d0321e1f0b","source":{"kind":"arxiv","id":"2507.15833","version":3},"attestation_state":"computed","paper":{"title":"Look, Focus, Act: Efficient and Robust Robot Learning via Human Gaze and Foveated Vision Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.RO","authors_text":"Andrew Lee, Dechen Gao, Ian Chuang, Iman Soltani, Jinyu Zou","submitted_at":"2025-07-21T17:44:10Z","abstract_excerpt":"Human vision is a highly active process driven by gaze, which directs attention to task-relevant regions through foveation, dramatically reducing visual processing. In contrast, robot learning systems typically rely on passive, uniform processing of raw camera images. In this work, we explore how incorporating human-like active gaze into robotic policies can enhance efficiency and robustness. We develop GIAVA (Gaze Integrated Active-Vision ALOHA), a robot vision system that emulates human head and neck movement, and gaze adjustment for foveated processing. Extending the AV-ALOHA robot platform"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.15833","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-07-21T17:44:10Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"de4d1c95e5283538660157a534f27f73ae1721fb9191b9d4c79ba460a83245eb","abstract_canon_sha256":"f07935476b44f83320ea1cb1439533cb62317d9a04575fb82759d94db96ddcc9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-15T00:21:12.350524Z","signature_b64":"SCj1NOxJ/f8LLc2PbkqFvs7EKV3bwtokyDH1w46QevxdV9OVYXE1OsphYyBTeOtWrHivLXHV0DKwm9g6HBHKAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eb1715c55e3e8542f615cc76edf0ebeefbd918649b754e7c0f96b5d0321e1f0b","last_reissued_at":"2026-07-15T00:21:12.349531Z","signature_status":"signed_v1","first_computed_at":"2026-07-15T00:21:12.349531Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Look, Focus, Act: Efficient and Robust Robot Learning via Human Gaze and Foveated Vision Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.RO","authors_text":"Andrew Lee, Dechen Gao, Ian Chuang, Iman Soltani, Jinyu Zou","submitted_at":"2025-07-21T17:44:10Z","abstract_excerpt":"Human vision is a highly active process driven by gaze, which directs attention to task-relevant regions through foveation, dramatically reducing visual processing. In contrast, robot learning systems typically rely on passive, uniform processing of raw camera images. In this work, we explore how incorporating human-like active gaze into robotic policies can enhance efficiency and robustness. We develop GIAVA (Gaze Integrated Active-Vision ALOHA), a robot vision system that emulates human head and neck movement, and gaze adjustment for foveated processing. Extending the AV-ALOHA robot platform"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.15833","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.15833/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.15833","created_at":"2026-07-15T00:21:12.349991+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.15833v3","created_at":"2026-07-15T00:21:12.349991+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.15833","created_at":"2026-07-15T00:21:12.349991+00:00"},{"alias_kind":"pith_short_12","alias_value":"5MLRLRK6H2CU","created_at":"2026-07-15T00:21:12.349991+00:00"},{"alias_kind":"pith_short_16","alias_value":"5MLRLRK6H2CUF5QV","created_at":"2026-07-15T00:21:12.349991+00:00"},{"alias_kind":"pith_short_8","alias_value":"5MLRLRK6","created_at":"2026-07-15T00:21:12.349991+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":4,"sample":[{"citing_arxiv_id":"2606.02565","citing_title":"Policy-based Foveated Imaging and Perception","ref_index":179,"is_internal_anchor":true},{"citing_arxiv_id":"2603.03243","citing_title":"HoMMI: Learning Whole-Body Mobile Manipulation from Human Demonstrations","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2604.22615","citing_title":"GazeVLA: Learning Human Intention for Robotic Manipulation","ref_index":19,"is_internal_anchor":true},{"citing_arxiv_id":"2604.04439","citing_title":"Estimating Central, Peripheral, and Temporal Visual Contributions to Human Decision Making in Atari Games","ref_index":10,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5MLRLRK6H2CUF5QVZR3O34HL53","json":"https://pith.science/pith/5MLRLRK6H2CUF5QVZR3O34HL53.json","graph_json":"https://pith.science/api/pith-number/5MLRLRK6H2CUF5QVZR3O34HL53/graph.json","events_json":"https://pith.science/api/pith-number/5MLRLRK6H2CUF5QVZR3O34HL53/events.json","paper":"https://pith.science/paper/5MLRLRK6"},"agent_actions":{"view_html":"https://pith.science/pith/5MLRLRK6H2CUF5QVZR3O34HL53","download_json":"https://pith.science/pith/5MLRLRK6H2CUF5QVZR3O34HL53.json","view_paper":"https://pith.science/paper/5MLRLRK6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.15833&json=true","fetch_graph":"https://pith.science/api/pith-number/5MLRLRK6H2CUF5QVZR3O34HL53/graph.json","fetch_events":"https://pith.science/api/pith-number/5MLRLRK6H2CUF5QVZR3O34HL53/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5MLRLRK6H2CUF5QVZR3O34HL53/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5MLRLRK6H2CUF5QVZR3O34HL53/action/storage_attestation","attest_author":"https://pith.science/pith/5MLRLRK6H2CUF5QVZR3O34HL53/action/author_attestation","sign_citation":"https://pith.science/pith/5MLRLRK6H2CUF5QVZR3O34HL53/action/citation_signature","submit_replication":"https://pith.science/pith/5MLRLRK6H2CUF5QVZR3O34HL53/action/replication_record"}},"created_at":"2026-07-15T00:21:12.349991+00:00","updated_at":"2026-07-15T00:21:12.349991+00:00"}