{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:VBLI3MOG272Z76OVVIFSFAFFCS","short_pith_number":"pith:VBLI3MOG","schema_version":"1.0","canonical_sha256":"a8568db1c6d7f59ff9d5aa0b2280a5148600a7567e541a2a9dce327d5d2885d3","source":{"kind":"arxiv","id":"2301.00794","version":3},"attestation_state":"computed","paper":{"title":"STEPs: Self-Supervised Key Step Extraction and Localization from Unlabeled Procedural Videos","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Anshul Shah, Benjamin Lundell, Harpreet Sawhney, Rama Chellappa","submitted_at":"2023-01-02T18:32:45Z","abstract_excerpt":"We address the problem of extracting key steps from unlabeled procedural videos, motivated by the potential of Augmented Reality (AR) headsets to revolutionize job training and performance. We decompose the problem into two steps: representation learning and key steps extraction. We propose a training objective, Bootstrapped Multi-Cue Contrastive (BMC2) loss to learn discriminative representations for various steps without any labels. Different from prior works, we develop techniques to train a light-weight temporal module which uses off-the-shelf features for self supervision. Our approach ca"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.00794","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-01-02T18:32:45Z","cross_cats_sorted":[],"title_canon_sha256":"f53a2a05e575ba411098825e49caf5a53b7b1c8f670714a04c3f7ee206ab427b","abstract_canon_sha256":"4384c40758ba8a8db1f8a498c21b1dbc90c452728192f15931775605dd577646"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:49:18.236548Z","signature_b64":"psaHBSmf3n19dTY73N95W9KGnzLoBRyFwDVsBu4eF2tEn3IV39bqduf0o7zMJUDCMimjNOMhBMy93XU/bdfTAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a8568db1c6d7f59ff9d5aa0b2280a5148600a7567e541a2a9dce327d5d2885d3","last_reissued_at":"2026-07-05T06:49:18.236054Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:49:18.236054Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"STEPs: Self-Supervised Key Step Extraction and Localization from Unlabeled Procedural Videos","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Anshul Shah, Benjamin Lundell, Harpreet Sawhney, Rama Chellappa","submitted_at":"2023-01-02T18:32:45Z","abstract_excerpt":"We address the problem of extracting key steps from unlabeled procedural videos, motivated by the potential of Augmented Reality (AR) headsets to revolutionize job training and performance. We decompose the problem into two steps: representation learning and key steps extraction. We propose a training objective, Bootstrapped Multi-Cue Contrastive (BMC2) loss to learn discriminative representations for various steps without any labels. Different from prior works, we develop techniques to train a light-weight temporal module which uses off-the-shelf features for self supervision. Our approach ca"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.00794","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.00794/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.00794","created_at":"2026-07-05T06:49:18.236110+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.00794v3","created_at":"2026-07-05T06:49:18.236110+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.00794","created_at":"2026-07-05T06:49:18.236110+00:00"},{"alias_kind":"pith_short_12","alias_value":"VBLI3MOG272Z","created_at":"2026-07-05T06:49:18.236110+00:00"},{"alias_kind":"pith_short_16","alias_value":"VBLI3MOG272Z76OV","created_at":"2026-07-05T06:49:18.236110+00:00"},{"alias_kind":"pith_short_8","alias_value":"VBLI3MOG","created_at":"2026-07-05T06:49:18.236110+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.11999","citing_title":"Generative Representational Learning of Foundation Models for Recommendation","ref_index":52,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VBLI3MOG272Z76OVVIFSFAFFCS","json":"https://pith.science/pith/VBLI3MOG272Z76OVVIFSFAFFCS.json","graph_json":"https://pith.science/api/pith-number/VBLI3MOG272Z76OVVIFSFAFFCS/graph.json","events_json":"https://pith.science/api/pith-number/VBLI3MOG272Z76OVVIFSFAFFCS/events.json","paper":"https://pith.science/paper/VBLI3MOG"},"agent_actions":{"view_html":"https://pith.science/pith/VBLI3MOG272Z76OVVIFSFAFFCS","download_json":"https://pith.science/pith/VBLI3MOG272Z76OVVIFSFAFFCS.json","view_paper":"https://pith.science/paper/VBLI3MOG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.00794&json=true","fetch_graph":"https://pith.science/api/pith-number/VBLI3MOG272Z76OVVIFSFAFFCS/graph.json","fetch_events":"https://pith.science/api/pith-number/VBLI3MOG272Z76OVVIFSFAFFCS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VBLI3MOG272Z76OVVIFSFAFFCS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VBLI3MOG272Z76OVVIFSFAFFCS/action/storage_attestation","attest_author":"https://pith.science/pith/VBLI3MOG272Z76OVVIFSFAFFCS/action/author_attestation","sign_citation":"https://pith.science/pith/VBLI3MOG272Z76OVVIFSFAFFCS/action/citation_signature","submit_replication":"https://pith.science/pith/VBLI3MOG272Z76OVVIFSFAFFCS/action/replication_record"}},"created_at":"2026-07-05T06:49:18.236110+00:00","updated_at":"2026-07-05T06:49:18.236110+00:00"}