{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:UGJSF7IICAVTN6LLB5ZGXRNCS7","short_pith_number":"pith:UGJSF7II","schema_version":"1.0","canonical_sha256":"a19322fd08102b36f96b0f726bc5a297fd2d6249b4c98cb62ad9afda00ac9108","source":{"kind":"arxiv","id":"2203.15187","version":1},"attestation_state":"computed","paper":{"title":"ASM-Loc: Action-aware Segment Modeling for Weakly-Supervised Temporal Action Localization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Abhinav Shrivastava, Bo He, Le Kang, Xin Zhou, Xitong Yang, Zhiyu Cheng","submitted_at":"2022-03-29T01:59:26Z","abstract_excerpt":"Weakly-supervised temporal action localization aims to recognize and localize action segments in untrimmed videos given only video-level action labels for training. Without the boundary information of action segments, existing methods mostly rely on multiple instance learning (MIL), where the predictions of unlabeled instances (i.e., video snippets) are supervised by classifying labeled bags (i.e., untrimmed videos). However, this formulation typically treats snippets in a video as independent instances, ignoring the underlying temporal structures within and across action segments. To address "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.15187","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-03-29T01:59:26Z","cross_cats_sorted":[],"title_canon_sha256":"83902881d503cb035900513aae256b84a68538fd29f401c552e33a627b91b16a","abstract_canon_sha256":"ada5d1a1ec08986d7da4f4bf31ef20f916a5c900c413d1293e33c0e338899553"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:09:22.600558Z","signature_b64":"bVcNb8U/H7guSdtTerapOghDf9dF58AQur0rQUvTPuNOgxDREMcMnVOSGFwYqsAgoeDWPIEX7BTGmaM0ZjYSDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a19322fd08102b36f96b0f726bc5a297fd2d6249b4c98cb62ad9afda00ac9108","last_reissued_at":"2026-07-05T04:09:22.600063Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:09:22.600063Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ASM-Loc: Action-aware Segment Modeling for Weakly-Supervised Temporal Action Localization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Abhinav Shrivastava, Bo He, Le Kang, Xin Zhou, Xitong Yang, Zhiyu Cheng","submitted_at":"2022-03-29T01:59:26Z","abstract_excerpt":"Weakly-supervised temporal action localization aims to recognize and localize action segments in untrimmed videos given only video-level action labels for training. Without the boundary information of action segments, existing methods mostly rely on multiple instance learning (MIL), where the predictions of unlabeled instances (i.e., video snippets) are supervised by classifying labeled bags (i.e., untrimmed videos). However, this formulation typically treats snippets in a video as independent instances, ignoring the underlying temporal structures within and across action segments. To address "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.15187","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.15187/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.15187","created_at":"2026-07-05T04:09:22.600126+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.15187v1","created_at":"2026-07-05T04:09:22.600126+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.15187","created_at":"2026-07-05T04:09:22.600126+00:00"},{"alias_kind":"pith_short_12","alias_value":"UGJSF7IICAVT","created_at":"2026-07-05T04:09:22.600126+00:00"},{"alias_kind":"pith_short_16","alias_value":"UGJSF7IICAVTN6LL","created_at":"2026-07-05T04:09:22.600126+00:00"},{"alias_kind":"pith_short_8","alias_value":"UGJSF7II","created_at":"2026-07-05T04:09:22.600126+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.16258","citing_title":"ViFusion: In-Network Tensor Fusion for Scalable Video Feature Indexing","ref_index":2022,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UGJSF7IICAVTN6LLB5ZGXRNCS7","json":"https://pith.science/pith/UGJSF7IICAVTN6LLB5ZGXRNCS7.json","graph_json":"https://pith.science/api/pith-number/UGJSF7IICAVTN6LLB5ZGXRNCS7/graph.json","events_json":"https://pith.science/api/pith-number/UGJSF7IICAVTN6LLB5ZGXRNCS7/events.json","paper":"https://pith.science/paper/UGJSF7II"},"agent_actions":{"view_html":"https://pith.science/pith/UGJSF7IICAVTN6LLB5ZGXRNCS7","download_json":"https://pith.science/pith/UGJSF7IICAVTN6LLB5ZGXRNCS7.json","view_paper":"https://pith.science/paper/UGJSF7II","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.15187&json=true","fetch_graph":"https://pith.science/api/pith-number/UGJSF7IICAVTN6LLB5ZGXRNCS7/graph.json","fetch_events":"https://pith.science/api/pith-number/UGJSF7IICAVTN6LLB5ZGXRNCS7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UGJSF7IICAVTN6LLB5ZGXRNCS7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UGJSF7IICAVTN6LLB5ZGXRNCS7/action/storage_attestation","attest_author":"https://pith.science/pith/UGJSF7IICAVTN6LLB5ZGXRNCS7/action/author_attestation","sign_citation":"https://pith.science/pith/UGJSF7IICAVTN6LLB5ZGXRNCS7/action/citation_signature","submit_replication":"https://pith.science/pith/UGJSF7IICAVTN6LLB5ZGXRNCS7/action/replication_record"}},"created_at":"2026-07-05T04:09:22.600126+00:00","updated_at":"2026-07-05T04:09:22.600126+00:00"}