{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:ZFLCIIRCDKX7ZF5ULXTIKTTDHB","short_pith_number":"pith:ZFLCIIRC","schema_version":"1.0","canonical_sha256":"c9562422221aaffc97b45de6854e63387ff405967d59ac02bba617d2f0b3cd51","source":{"kind":"arxiv","id":"2208.04464","version":3},"attestation_state":"computed","paper":{"title":"In the Eye of Transformer: Global-Local Correlation for Egocentric Gaze Estimation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bolin Lai, Fiona Ryan, James M. Rehg, Miao Liu","submitted_at":"2022-08-08T23:25:05Z","abstract_excerpt":"In this paper, we present the first transformer-based model to address the challenging problem of egocentric gaze estimation. We observe that the connection between the global scene context and local visual information is vital for localizing the gaze fixation from egocentric video frames. To this end, we design the transformer encoder to embed the global context as one additional visual token and further propose a novel Global-Local Correlation (GLC) module to explicitly model the correlation of the global token and each local token. We validate our model on two egocentric video datasets - EG"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2208.04464","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2022-08-08T23:25:05Z","cross_cats_sorted":[],"title_canon_sha256":"153151121d3728e8ff4eff049845ae7345d1c7ac20a2a8070e2ef91f1bc2be9b","abstract_canon_sha256":"a13e9874aa6fef4b07ab63e3f2638d2a6a8b38bdcb08ffa64f9dcc4fa65310f6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:21:00.109596Z","signature_b64":"My/YhZpcnJLicSU0MsT9tcZ9DWQV6SqOQd1XQEybtB7yRwPZlKJJkMxI0w9kwVs/Lc2RXW4U7ROV86EDlP01Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c9562422221aaffc97b45de6854e63387ff405967d59ac02bba617d2f0b3cd51","last_reissued_at":"2026-07-05T09:21:00.109001Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:21:00.109001Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"In the Eye of Transformer: Global-Local Correlation for Egocentric Gaze Estimation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bolin Lai, Fiona Ryan, James M. Rehg, Miao Liu","submitted_at":"2022-08-08T23:25:05Z","abstract_excerpt":"In this paper, we present the first transformer-based model to address the challenging problem of egocentric gaze estimation. We observe that the connection between the global scene context and local visual information is vital for localizing the gaze fixation from egocentric video frames. To this end, we design the transformer encoder to embed the global context as one additional visual token and further propose a novel Global-Local Correlation (GLC) module to explicitly model the correlation of the global token and each local token. We validate our model on two egocentric video datasets - EG"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2208.04464","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2208.04464/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2208.04464","created_at":"2026-07-05T09:21:00.109072+00:00"},{"alias_kind":"arxiv_version","alias_value":"2208.04464v3","created_at":"2026-07-05T09:21:00.109072+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2208.04464","created_at":"2026-07-05T09:21:00.109072+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZFLCIIRCDKX7","created_at":"2026-07-05T09:21:00.109072+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZFLCIIRCDKX7ZF5U","created_at":"2026-07-05T09:21:00.109072+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZFLCIIRC","created_at":"2026-07-05T09:21:00.109072+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.19629","citing_title":"SkillSight: Efficient First-Person Skill Assessment with Gaze","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07859","citing_title":"EyeCue: Driver Cognitive Distraction Detection via Gaze-Empowered Egocentric Video Understanding","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZFLCIIRCDKX7ZF5ULXTIKTTDHB","json":"https://pith.science/pith/ZFLCIIRCDKX7ZF5ULXTIKTTDHB.json","graph_json":"https://pith.science/api/pith-number/ZFLCIIRCDKX7ZF5ULXTIKTTDHB/graph.json","events_json":"https://pith.science/api/pith-number/ZFLCIIRCDKX7ZF5ULXTIKTTDHB/events.json","paper":"https://pith.science/paper/ZFLCIIRC"},"agent_actions":{"view_html":"https://pith.science/pith/ZFLCIIRCDKX7ZF5ULXTIKTTDHB","download_json":"https://pith.science/pith/ZFLCIIRCDKX7ZF5ULXTIKTTDHB.json","view_paper":"https://pith.science/paper/ZFLCIIRC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2208.04464&json=true","fetch_graph":"https://pith.science/api/pith-number/ZFLCIIRCDKX7ZF5ULXTIKTTDHB/graph.json","fetch_events":"https://pith.science/api/pith-number/ZFLCIIRCDKX7ZF5ULXTIKTTDHB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZFLCIIRCDKX7ZF5ULXTIKTTDHB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZFLCIIRCDKX7ZF5ULXTIKTTDHB/action/storage_attestation","attest_author":"https://pith.science/pith/ZFLCIIRCDKX7ZF5ULXTIKTTDHB/action/author_attestation","sign_citation":"https://pith.science/pith/ZFLCIIRCDKX7ZF5ULXTIKTTDHB/action/citation_signature","submit_replication":"https://pith.science/pith/ZFLCIIRCDKX7ZF5ULXTIKTTDHB/action/replication_record"}},"created_at":"2026-07-05T09:21:00.109072+00:00","updated_at":"2026-07-05T09:21:00.109072+00:00"}