{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:T2KWKCMINRHIPH2YXTSR6PXH4X","short_pith_number":"pith:T2KWKCMI","schema_version":"1.0","canonical_sha256":"9e956509886c4e879f58bce51f3ee7e5f0b2ffd4e83459c2e0ab051b48c5ddd4","source":{"kind":"arxiv","id":"2405.17719","version":3},"attestation_state":"computed","paper":{"title":"Do Egocentric Video-Language Models Truly Understand Hand-Object Interactions?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Boshen Xu, Qin Jin, Sipeng Zheng, Yang Du, Zhinan Song, Ziheng Wang","submitted_at":"2024-05-28T00:27:29Z","abstract_excerpt":"Egocentric video-language pretraining is a crucial step in advancing the understanding of hand-object interactions in first-person scenarios. Despite successes on existing testbeds, we find that current EgoVLMs can be easily misled by simple modifications, such as changing the verbs or nouns in interaction descriptions, with models struggling to distinguish between these changes. This raises the question: Do EgoVLMs truly understand hand-object interactions? To address this question, we introduce a benchmark called EgoHOIBench, revealing the performance limitation of current egocentric models "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.17719","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-28T00:27:29Z","cross_cats_sorted":[],"title_canon_sha256":"e2dff371ad669fea27f1f9c197b523333b33f1d9a6aa68ec3f003a9773f00a48","abstract_canon_sha256":"33710da22e09bf22fe34185b82fcc9e933105ab1343a3a8fc1a8104eb6f868c9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:08.476584Z","signature_b64":"rarSDKvsNAsBv1A2AtBK+uAk+F3IYxb6uTJ6m00OG2wHDc9rtz9P+Rp1NGHXTxDOTL26OBBHz6Q1Cxri/QH6AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9e956509886c4e879f58bce51f3ee7e5f0b2ffd4e83459c2e0ab051b48c5ddd4","last_reissued_at":"2026-07-05T10:17:08.476116Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:08.476116Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do Egocentric Video-Language Models Truly Understand Hand-Object Interactions?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Boshen Xu, Qin Jin, Sipeng Zheng, Yang Du, Zhinan Song, Ziheng Wang","submitted_at":"2024-05-28T00:27:29Z","abstract_excerpt":"Egocentric video-language pretraining is a crucial step in advancing the understanding of hand-object interactions in first-person scenarios. Despite successes on existing testbeds, we find that current EgoVLMs can be easily misled by simple modifications, such as changing the verbs or nouns in interaction descriptions, with models struggling to distinguish between these changes. This raises the question: Do EgoVLMs truly understand hand-object interactions? To address this question, we introduce a benchmark called EgoHOIBench, revealing the performance limitation of current egocentric models "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.17719","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.17719/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.17719","created_at":"2026-07-05T10:17:08.476169+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.17719v3","created_at":"2026-07-05T10:17:08.476169+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.17719","created_at":"2026-07-05T10:17:08.476169+00:00"},{"alias_kind":"pith_short_12","alias_value":"T2KWKCMINRHI","created_at":"2026-07-05T10:17:08.476169+00:00"},{"alias_kind":"pith_short_16","alias_value":"T2KWKCMINRHIPH2Y","created_at":"2026-07-05T10:17:08.476169+00:00"},{"alias_kind":"pith_short_8","alias_value":"T2KWKCMI","created_at":"2026-07-05T10:17:08.476169+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08514","citing_title":"Do Egocentric Video-Language Models Capture Both Hand- and Object-Centric Cues?","ref_index":34,"is_internal_anchor":true},{"citing_arxiv_id":"2604.10517","citing_title":"From Perception to Planning: Evolving Ego-Centric Task-Oriented Spatiotemporal Reasoning via Curriculum Learning","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T2KWKCMINRHIPH2YXTSR6PXH4X","json":"https://pith.science/pith/T2KWKCMINRHIPH2YXTSR6PXH4X.json","graph_json":"https://pith.science/api/pith-number/T2KWKCMINRHIPH2YXTSR6PXH4X/graph.json","events_json":"https://pith.science/api/pith-number/T2KWKCMINRHIPH2YXTSR6PXH4X/events.json","paper":"https://pith.science/paper/T2KWKCMI"},"agent_actions":{"view_html":"https://pith.science/pith/T2KWKCMINRHIPH2YXTSR6PXH4X","download_json":"https://pith.science/pith/T2KWKCMINRHIPH2YXTSR6PXH4X.json","view_paper":"https://pith.science/paper/T2KWKCMI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.17719&json=true","fetch_graph":"https://pith.science/api/pith-number/T2KWKCMINRHIPH2YXTSR6PXH4X/graph.json","fetch_events":"https://pith.science/api/pith-number/T2KWKCMINRHIPH2YXTSR6PXH4X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T2KWKCMINRHIPH2YXTSR6PXH4X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T2KWKCMINRHIPH2YXTSR6PXH4X/action/storage_attestation","attest_author":"https://pith.science/pith/T2KWKCMINRHIPH2YXTSR6PXH4X/action/author_attestation","sign_citation":"https://pith.science/pith/T2KWKCMINRHIPH2YXTSR6PXH4X/action/citation_signature","submit_replication":"https://pith.science/pith/T2KWKCMINRHIPH2YXTSR6PXH4X/action/replication_record"}},"created_at":"2026-07-05T10:17:08.476169+00:00","updated_at":"2026-07-05T10:17:08.476169+00:00"}