{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TVLZPACYUZDXCEUKVPYTYI6ER2","short_pith_number":"pith:TVLZPACY","schema_version":"1.0","canonical_sha256":"9d57978058a64771128aabf13c23c48e9450cbdf8ecb8cb8d3483341143f0b9c","source":{"kind":"arxiv","id":"2406.18070","version":4},"attestation_state":"computed","paper":{"title":"EgoVideo: Exploring Egocentric Foundation Model and Downstream Adaptation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Baoqi Pei, Guo Chen, Jilan Xu, Kanghua Pan, Limin Wang, Tong Lu, Yali Wang, Yicheng Liu, Yifei Huang, Yuping He, Yu Qiao","submitted_at":"2024-06-26T05:01:37Z","abstract_excerpt":"In this report, we present our solutions to the EgoVis Challenges in CVPR 2024, including five tracks in the Ego4D challenge and three tracks in the EPIC-Kitchens challenge. Building upon the video-language two-tower model and leveraging our meticulously organized egocentric video data, we introduce a novel foundation model called EgoVideo. This model is specifically designed to cater to the unique characteristics of egocentric videos and provides strong support for our competition submissions. In the Ego4D challenges, we tackle various tasks including Natural Language Queries, Step Grounding,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.18070","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-26T05:01:37Z","cross_cats_sorted":[],"title_canon_sha256":"7398db48055822ec51bf5bd815e4060198d053b2708c93d411620f043043e204","abstract_canon_sha256":"1e97c94c117109a63d5bffd286b538494f13902db3cde5f99d77b27f25e3c13b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:38:12.157297Z","signature_b64":"EH3WKL4SplWjA/IHUxpvlqzJvHGkvYPbb4/q0LL/UHHnOg/h84HrexCMmkOZAvcRlvirMamwhs+dLUt0AHv2Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9d57978058a64771128aabf13c23c48e9450cbdf8ecb8cb8d3483341143f0b9c","last_reissued_at":"2026-07-05T08:38:12.156832Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:38:12.156832Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EgoVideo: Exploring Egocentric Foundation Model and Downstream Adaptation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Baoqi Pei, Guo Chen, Jilan Xu, Kanghua Pan, Limin Wang, Tong Lu, Yali Wang, Yicheng Liu, Yifei Huang, Yuping He, Yu Qiao","submitted_at":"2024-06-26T05:01:37Z","abstract_excerpt":"In this report, we present our solutions to the EgoVis Challenges in CVPR 2024, including five tracks in the Ego4D challenge and three tracks in the EPIC-Kitchens challenge. Building upon the video-language two-tower model and leveraging our meticulously organized egocentric video data, we introduce a novel foundation model called EgoVideo. This model is specifically designed to cater to the unique characteristics of egocentric videos and provides strong support for our competition submissions. In the Ego4D challenges, we tackle various tasks including Natural Language Queries, Step Grounding,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.18070","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.18070/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.18070","created_at":"2026-07-05T08:38:12.156896+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.18070v4","created_at":"2026-07-05T08:38:12.156896+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.18070","created_at":"2026-07-05T08:38:12.156896+00:00"},{"alias_kind":"pith_short_12","alias_value":"TVLZPACYUZDX","created_at":"2026-07-05T08:38:12.156896+00:00"},{"alias_kind":"pith_short_16","alias_value":"TVLZPACYUZDXCEUK","created_at":"2026-07-05T08:38:12.156896+00:00"},{"alias_kind":"pith_short_8","alias_value":"TVLZPACY","created_at":"2026-07-05T08:38:12.156896+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09142","citing_title":"Decoding Pedestrian Crossing Intention from Egocentric Vision via Vision Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31227","citing_title":"HiERO-StepG @ Ego4D Step Grounding Challenge: hierarchical activity understanding enables zero-shot step grounding","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24893","citing_title":"Interactive Episodic Memory with User Feedback","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07642","citing_title":"EggHand: A Multimodal Foundation Model for Egocentric Hand Pose Forecasting","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07859","citing_title":"EyeCue: Driver Cognitive Distraction Detection via Gaze-Empowered Egocentric Video Understanding","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TVLZPACYUZDXCEUKVPYTYI6ER2","json":"https://pith.science/pith/TVLZPACYUZDXCEUKVPYTYI6ER2.json","graph_json":"https://pith.science/api/pith-number/TVLZPACYUZDXCEUKVPYTYI6ER2/graph.json","events_json":"https://pith.science/api/pith-number/TVLZPACYUZDXCEUKVPYTYI6ER2/events.json","paper":"https://pith.science/paper/TVLZPACY"},"agent_actions":{"view_html":"https://pith.science/pith/TVLZPACYUZDXCEUKVPYTYI6ER2","download_json":"https://pith.science/pith/TVLZPACYUZDXCEUKVPYTYI6ER2.json","view_paper":"https://pith.science/paper/TVLZPACY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.18070&json=true","fetch_graph":"https://pith.science/api/pith-number/TVLZPACYUZDXCEUKVPYTYI6ER2/graph.json","fetch_events":"https://pith.science/api/pith-number/TVLZPACYUZDXCEUKVPYTYI6ER2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TVLZPACYUZDXCEUKVPYTYI6ER2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TVLZPACYUZDXCEUKVPYTYI6ER2/action/storage_attestation","attest_author":"https://pith.science/pith/TVLZPACYUZDXCEUKVPYTYI6ER2/action/author_attestation","sign_citation":"https://pith.science/pith/TVLZPACYUZDXCEUKVPYTYI6ER2/action/citation_signature","submit_replication":"https://pith.science/pith/TVLZPACYUZDXCEUKVPYTYI6ER2/action/replication_record"}},"created_at":"2026-07-05T08:38:12.156896+00:00","updated_at":"2026-07-05T08:38:12.156896+00:00"}