{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FKP53OALSLNZ4OW6S6F3STCXAV","short_pith_number":"pith:FKP53OAL","schema_version":"1.0","canonical_sha256":"2a9fddb80b92db9e3ade978bb94c57054fa0ad8db6928b540ba99b821a0691ee","source":{"kind":"arxiv","id":"2502.17352","version":1},"attestation_state":"computed","paper":{"title":"Leveraging Procedural Knowledge and Task Hierarchies for Efficient Instructional Video Pre-training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Irfan Essa, Karan Samel, Nitish Sontakke","submitted_at":"2025-02-24T17:29:10Z","abstract_excerpt":"Instructional videos provide a convenient modality to learn new tasks (ex. cooking a recipe, or assembling furniture). A viewer will want to find a corresponding video that reflects both the overall task they are interested in as well as contains the relevant steps they need to carry out the task. To perform this, an instructional video model should be capable of inferring both the tasks and the steps that occur in an input video. Doing this efficiently and in a generalizable fashion is key when compute or relevant video topics used to train this model are limited. To address these requirement"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.17352","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-24T17:29:10Z","cross_cats_sorted":[],"title_canon_sha256":"07db2ebbfac4c7223f39753441d0ff6f6f408298deb66ae95390d08c9fab04b9","abstract_canon_sha256":"3a7859be0205368746eeb31b28598a853baaa9f9021b60848c9fba1041e59942"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:09.300555Z","signature_b64":"OnQu60w3RnjwxcloPPqohgP+3DQTPjiMpyijX844VUtTQGUK4gBAjplo6QOC+kf/gX88WCUnt+yKFy4VDVZdBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2a9fddb80b92db9e3ade978bb94c57054fa0ad8db6928b540ba99b821a0691ee","last_reissued_at":"2026-07-05T10:19:09.300074Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:09.300074Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Leveraging Procedural Knowledge and Task Hierarchies for Efficient Instructional Video Pre-training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Irfan Essa, Karan Samel, Nitish Sontakke","submitted_at":"2025-02-24T17:29:10Z","abstract_excerpt":"Instructional videos provide a convenient modality to learn new tasks (ex. cooking a recipe, or assembling furniture). A viewer will want to find a corresponding video that reflects both the overall task they are interested in as well as contains the relevant steps they need to carry out the task. To perform this, an instructional video model should be capable of inferring both the tasks and the steps that occur in an input video. Doing this efficiently and in a generalizable fashion is key when compute or relevant video topics used to train this model are limited. To address these requirement"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.17352","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.17352/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.17352","created_at":"2026-07-05T10:19:09.300138+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.17352v1","created_at":"2026-07-05T10:19:09.300138+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.17352","created_at":"2026-07-05T10:19:09.300138+00:00"},{"alias_kind":"pith_short_12","alias_value":"FKP53OALSLNZ","created_at":"2026-07-05T10:19:09.300138+00:00"},{"alias_kind":"pith_short_16","alias_value":"FKP53OALSLNZ4OW6","created_at":"2026-07-05T10:19:09.300138+00:00"},{"alias_kind":"pith_short_8","alias_value":"FKP53OAL","created_at":"2026-07-05T10:19:09.300138+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.16425","citing_title":"$I^2G$: Generating Instructional Illustrations via Text-Conditioned Diffusion","ref_index":39,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FKP53OALSLNZ4OW6S6F3STCXAV","json":"https://pith.science/pith/FKP53OALSLNZ4OW6S6F3STCXAV.json","graph_json":"https://pith.science/api/pith-number/FKP53OALSLNZ4OW6S6F3STCXAV/graph.json","events_json":"https://pith.science/api/pith-number/FKP53OALSLNZ4OW6S6F3STCXAV/events.json","paper":"https://pith.science/paper/FKP53OAL"},"agent_actions":{"view_html":"https://pith.science/pith/FKP53OALSLNZ4OW6S6F3STCXAV","download_json":"https://pith.science/pith/FKP53OALSLNZ4OW6S6F3STCXAV.json","view_paper":"https://pith.science/paper/FKP53OAL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.17352&json=true","fetch_graph":"https://pith.science/api/pith-number/FKP53OALSLNZ4OW6S6F3STCXAV/graph.json","fetch_events":"https://pith.science/api/pith-number/FKP53OALSLNZ4OW6S6F3STCXAV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FKP53OALSLNZ4OW6S6F3STCXAV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FKP53OALSLNZ4OW6S6F3STCXAV/action/storage_attestation","attest_author":"https://pith.science/pith/FKP53OALSLNZ4OW6S6F3STCXAV/action/author_attestation","sign_citation":"https://pith.science/pith/FKP53OALSLNZ4OW6S6F3STCXAV/action/citation_signature","submit_replication":"https://pith.science/pith/FKP53OALSLNZ4OW6S6F3STCXAV/action/replication_record"}},"created_at":"2026-07-05T10:19:09.300138+00:00","updated_at":"2026-07-05T10:19:09.300138+00:00"}