{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TWGNFPZJIZVN2YP33CIJSFMULQ","short_pith_number":"pith:TWGNFPZJ","schema_version":"1.0","canonical_sha256":"9d8cd2bf29466add61fbd8909915945c3ff45ef8e7065bbde3ee86793788fc95","source":{"kind":"arxiv","id":"2511.18242","version":3},"attestation_state":"computed","paper":{"title":"EgoVITA: Learning to Plan and Verify for Egocentric Video Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Pooyan Fazli, Yogesh Kulkarni","submitted_at":"2025-11-23T01:25:17Z","abstract_excerpt":"Egocentric video understanding requires procedural reasoning under partial observability and continuously shifting viewpoints. Current multimodal large language models (MLLMs) struggle with this setting, often generating plausible but visually inconsistent or weakly grounded responses. We introduce $\\textbf{EgoVITA}$, a framework that decomposes egocentric video reasoning into a structured $\\textit{plan-then-verify}$ process. The model first generates an $\\textbf{egocentric plan}$: a causal sequence of anticipated actions from a first-person perspective. This plan is then evaluated by an $\\tex"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2511.18242","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-11-23T01:25:17Z","cross_cats_sorted":[],"title_canon_sha256":"ddfd0dd649c517d51573193deb45819d5ab72728b4eeb4828fbfc7f69a08713c","abstract_canon_sha256":"b5e32004b9bad0cebd772b7c4ca61bc2f6224b1a675c25f56df4ffb5386959cc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-01T01:17:11.900198Z","signature_b64":"sIingQK1xb0pFCN7Xl/P8ZbblJGN5TnR2r/qQ7KTHo2Ad/qCRHt73RAOQj/OU4A74qKrkh9xbDOmdGiPKaj+Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9d8cd2bf29466add61fbd8909915945c3ff45ef8e7065bbde3ee86793788fc95","last_reissued_at":"2026-07-01T01:17:11.899677Z","signature_status":"signed_v1","first_computed_at":"2026-07-01T01:17:11.899677Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EgoVITA: Learning to Plan and Verify for Egocentric Video Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Pooyan Fazli, Yogesh Kulkarni","submitted_at":"2025-11-23T01:25:17Z","abstract_excerpt":"Egocentric video understanding requires procedural reasoning under partial observability and continuously shifting viewpoints. Current multimodal large language models (MLLMs) struggle with this setting, often generating plausible but visually inconsistent or weakly grounded responses. We introduce $\\textbf{EgoVITA}$, a framework that decomposes egocentric video reasoning into a structured $\\textit{plan-then-verify}$ process. The model first generates an $\\textbf{egocentric plan}$: a causal sequence of anticipated actions from a first-person perspective. This plan is then evaluated by an $\\tex"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2511.18242","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2511.18242/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2511.18242","created_at":"2026-07-01T01:17:11.899742+00:00"},{"alias_kind":"arxiv_version","alias_value":"2511.18242v3","created_at":"2026-07-01T01:17:11.899742+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2511.18242","created_at":"2026-07-01T01:17:11.899742+00:00"},{"alias_kind":"pith_short_12","alias_value":"TWGNFPZJIZVN","created_at":"2026-07-01T01:17:11.899742+00:00"},{"alias_kind":"pith_short_16","alias_value":"TWGNFPZJIZVN2YP3","created_at":"2026-07-01T01:17:11.899742+00:00"},{"alias_kind":"pith_short_8","alias_value":"TWGNFPZJ","created_at":"2026-07-01T01:17:11.899742+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2605.18734","citing_title":"EgoExoMem: Cross-View Memory Reasoning over Synchronized Egocentric and Exocentric Videos","ref_index":44,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TWGNFPZJIZVN2YP33CIJSFMULQ","json":"https://pith.science/pith/TWGNFPZJIZVN2YP33CIJSFMULQ.json","graph_json":"https://pith.science/api/pith-number/TWGNFPZJIZVN2YP33CIJSFMULQ/graph.json","events_json":"https://pith.science/api/pith-number/TWGNFPZJIZVN2YP33CIJSFMULQ/events.json","paper":"https://pith.science/paper/TWGNFPZJ"},"agent_actions":{"view_html":"https://pith.science/pith/TWGNFPZJIZVN2YP33CIJSFMULQ","download_json":"https://pith.science/pith/TWGNFPZJIZVN2YP33CIJSFMULQ.json","view_paper":"https://pith.science/paper/TWGNFPZJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2511.18242&json=true","fetch_graph":"https://pith.science/api/pith-number/TWGNFPZJIZVN2YP33CIJSFMULQ/graph.json","fetch_events":"https://pith.science/api/pith-number/TWGNFPZJIZVN2YP33CIJSFMULQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TWGNFPZJIZVN2YP33CIJSFMULQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TWGNFPZJIZVN2YP33CIJSFMULQ/action/storage_attestation","attest_author":"https://pith.science/pith/TWGNFPZJIZVN2YP33CIJSFMULQ/action/author_attestation","sign_citation":"https://pith.science/pith/TWGNFPZJIZVN2YP33CIJSFMULQ/action/citation_signature","submit_replication":"https://pith.science/pith/TWGNFPZJIZVN2YP33CIJSFMULQ/action/replication_record"}},"created_at":"2026-07-01T01:17:11.899742+00:00","updated_at":"2026-07-01T01:17:11.899742+00:00"}