{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WPX6WVSUI2ZNLPZPEX4MDROFXK","short_pith_number":"pith:WPX6WVSU","schema_version":"1.0","canonical_sha256":"b3efeb565446b2d5bf2f25f8c1c5c5ba876d1093a0cc66c217a5bad4ce51fa58","source":{"kind":"arxiv","id":"2411.08466","version":2},"attestation_state":"computed","paper":{"title":"Weakly Supervised Temporal Action Localization via Dual-Prior Collaborative Learning Guided by Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chun Yuan, Jinwei Fang, Ke Zhang, Quan Zhang, Rui Yuan, Xi Tang, Yuxin Qi","submitted_at":"2024-11-13T09:37:24Z","abstract_excerpt":"Recent breakthroughs in Multimodal Large Language Models (MLLMs) have gained significant recognition within the deep learning community, where the fusion of the Video Foundation Models (VFMs) and Large Language Models(LLMs) has proven instrumental in constructing robust video understanding systems, effectively surmounting constraints associated with predefined visual tasks. These sophisticated MLLMs exhibit remarkable proficiency in comprehending videos, swiftly attaining unprecedented performance levels across diverse benchmarks. However, their operation demands substantial memory and computa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.08466","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-13T09:37:24Z","cross_cats_sorted":[],"title_canon_sha256":"9ecd6fd9995004f89b84f5be4e2f09e0e1f9b433d4b3dcbbb83453a90858b975","abstract_canon_sha256":"4faab02b79d0683f67351e6fc6797281dc305958ef81f121d5dacba2edb4ae34"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:15.795951Z","signature_b64":"sAp7c/vVSPq0n8+XierSGjHZCNdfXkWyE0NuJVPZ8ndYYlZk9IP/9b8AManmcbpvru9iQc8YHJ79YP+4P0OWBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b3efeb565446b2d5bf2f25f8c1c5c5ba876d1093a0cc66c217a5bad4ce51fa58","last_reissued_at":"2026-07-05T11:18:15.795446Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:15.795446Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Weakly Supervised Temporal Action Localization via Dual-Prior Collaborative Learning Guided by Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chun Yuan, Jinwei Fang, Ke Zhang, Quan Zhang, Rui Yuan, Xi Tang, Yuxin Qi","submitted_at":"2024-11-13T09:37:24Z","abstract_excerpt":"Recent breakthroughs in Multimodal Large Language Models (MLLMs) have gained significant recognition within the deep learning community, where the fusion of the Video Foundation Models (VFMs) and Large Language Models(LLMs) has proven instrumental in constructing robust video understanding systems, effectively surmounting constraints associated with predefined visual tasks. These sophisticated MLLMs exhibit remarkable proficiency in comprehending videos, swiftly attaining unprecedented performance levels across diverse benchmarks. However, their operation demands substantial memory and computa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.08466","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.08466/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.08466","created_at":"2026-07-05T11:18:15.795508+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.08466v2","created_at":"2026-07-05T11:18:15.795508+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.08466","created_at":"2026-07-05T11:18:15.795508+00:00"},{"alias_kind":"pith_short_12","alias_value":"WPX6WVSUI2ZN","created_at":"2026-07-05T11:18:15.795508+00:00"},{"alias_kind":"pith_short_16","alias_value":"WPX6WVSUI2ZNLPZP","created_at":"2026-07-05T11:18:15.795508+00:00"},{"alias_kind":"pith_short_8","alias_value":"WPX6WVSU","created_at":"2026-07-05T11:18:15.795508+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.17516","citing_title":"EASE: Embodied Active Event Perception via Self-Supervised Energy Minimization","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WPX6WVSUI2ZNLPZPEX4MDROFXK","json":"https://pith.science/pith/WPX6WVSUI2ZNLPZPEX4MDROFXK.json","graph_json":"https://pith.science/api/pith-number/WPX6WVSUI2ZNLPZPEX4MDROFXK/graph.json","events_json":"https://pith.science/api/pith-number/WPX6WVSUI2ZNLPZPEX4MDROFXK/events.json","paper":"https://pith.science/paper/WPX6WVSU"},"agent_actions":{"view_html":"https://pith.science/pith/WPX6WVSUI2ZNLPZPEX4MDROFXK","download_json":"https://pith.science/pith/WPX6WVSUI2ZNLPZPEX4MDROFXK.json","view_paper":"https://pith.science/paper/WPX6WVSU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.08466&json=true","fetch_graph":"https://pith.science/api/pith-number/WPX6WVSUI2ZNLPZPEX4MDROFXK/graph.json","fetch_events":"https://pith.science/api/pith-number/WPX6WVSUI2ZNLPZPEX4MDROFXK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WPX6WVSUI2ZNLPZPEX4MDROFXK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WPX6WVSUI2ZNLPZPEX4MDROFXK/action/storage_attestation","attest_author":"https://pith.science/pith/WPX6WVSUI2ZNLPZPEX4MDROFXK/action/author_attestation","sign_citation":"https://pith.science/pith/WPX6WVSUI2ZNLPZPEX4MDROFXK/action/citation_signature","submit_replication":"https://pith.science/pith/WPX6WVSUI2ZNLPZPEX4MDROFXK/action/replication_record"}},"created_at":"2026-07-05T11:18:15.795508+00:00","updated_at":"2026-07-05T11:18:15.795508+00:00"}