{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:FDXUCHFTMJF4H7XRHKYNLNHYA5","short_pith_number":"pith:FDXUCHFT","canonical_record":{"source":{"id":"2604.03329","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-04-02T22:14:25Z","cross_cats_sorted":["cs.AI","cs.LG","cs.SD"],"title_canon_sha256":"e8cab262b431ef21758c7550d53c6438f5b737844fd2fcdfb66a1b470757e399","abstract_canon_sha256":"92c57b3adfce891048610ac17661527dfcd3a71c3cb7e0c30b48c00e3a3b5a25"},"schema_version":"1.0"},"canonical_sha256":"28ef411cb3624bc3fef13ab0d5b4f8074249d28929eb08f580251bd43ce8ed0b","source":{"kind":"arxiv","id":"2604.03329","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.03329","created_at":"2026-07-07T02:18:39Z"},{"alias_kind":"arxiv_version","alias_value":"2604.03329v2","created_at":"2026-07-07T02:18:39Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.03329","created_at":"2026-07-07T02:18:39Z"},{"alias_kind":"pith_short_12","alias_value":"FDXUCHFTMJF4","created_at":"2026-07-07T02:18:39Z"},{"alias_kind":"pith_short_16","alias_value":"FDXUCHFTMJF4H7XR","created_at":"2026-07-07T02:18:39Z"},{"alias_kind":"pith_short_8","alias_value":"FDXUCHFT","created_at":"2026-07-07T02:18:39Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:FDXUCHFTMJF4H7XRHKYNLNHYA5","target":"record","payload":{"canonical_record":{"source":{"id":"2604.03329","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-04-02T22:14:25Z","cross_cats_sorted":["cs.AI","cs.LG","cs.SD"],"title_canon_sha256":"e8cab262b431ef21758c7550d53c6438f5b737844fd2fcdfb66a1b470757e399","abstract_canon_sha256":"92c57b3adfce891048610ac17661527dfcd3a71c3cb7e0c30b48c00e3a3b5a25"},"schema_version":"1.0"},"canonical_sha256":"28ef411cb3624bc3fef13ab0d5b4f8074249d28929eb08f580251bd43ce8ed0b","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-07T02:18:39.571046Z","signature_b64":"3+/5zx+SzvsmbhCpdCfZfuw1bE6FwIvKYlAlB1TeVAS1Wy0tCspXJtfgcpOS70BfKxjrhw739rellC5d2kwMCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"28ef411cb3624bc3fef13ab0d5b4f8074249d28929eb08f580251bd43ce8ed0b","last_reissued_at":"2026-07-07T02:18:39.570096Z","signature_status":"signed_v1","first_computed_at":"2026-07-07T02:18:39.570096Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2604.03329","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-07T02:18:39Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"5IAWVX0fMkQdV2ZO8/E5rqh545hSWYN1YUKovwive5UNcovI2c99uTn173uL5TfR8U9Esk3mtbE/3LMylmIGDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-20T10:55:59.641221Z"},"content_sha256":"25d797b5db537c095103a99f32e4cb0e57730c5c4bc4cbb574abe7a2a03e48fd","schema_version":"1.0","event_id":"sha256:25d797b5db537c095103a99f32e4cb0e57730c5c4bc4cbb574abe7a2a03e48fd"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:FDXUCHFTMJF4H7XRHKYNLNHYA5","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"AViS-Mamba: Adaptive Visual Steering of Audio State-Space Dynamics for Violence Detection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"Video CLS tokens steer audio Mamba parameters via conditional LoRA to improve multimodal violence detection accuracy.","cross_cats":["cs.AI","cs.LG","cs.SD"],"primary_cat":"cs.CV","authors_text":"Damith Chamalke Senadeera, Dimitrios Kollias, Gregory Slabaugh","submitted_at":"2026-04-02T22:14:25Z","abstract_excerpt":"Automatic violence detection from video is challenging because violent interactions may be distant, occluded, or only partially visible. Audio can provide complementary evidence for violent events that are difficult to recognize from visual information alone. However, audio itself may be absent, dubbed, or dominated by environmental noise, making the central challenge not whether to incorporate audio but how to adapt reliance on it according to the visual scene. We introduce \\emph{AViS-Mamba}, an audiovisual Mamba-based architecture in which the visual stream directly governs the behavior of t"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"On audio-filtered clip-level subsets of NTU-CCTV and DVD, CoLoRSMamba outperforms representative audio-only, video-only, and multimodal baselines, achieving 88.63% accuracy / 86.24% F1-V on NTU-CCTV and 75.77% accuracy / 72.94% F1-V on DVD while offering a favorable accuracy-efficiency tradeoff with fewer parameters and FLOPs.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the CLS token from VideoMamba can reliably generate channel-wise modulation vectors and stabilization gates that adapt AudioMamba's Delta, B, C parameters to produce genuinely scene-aware audio dynamics without introducing misalignment or noise in real-world conditions.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"CoLoRSMamba steers AudioMamba using video CLS-guided conditional LoRA to adapt selective state-space parameters, outperforming baselines on audio-filtered NTU-CCTV and DVD subsets with 88.63% and 75.77% accuracy respectively.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Video CLS tokens steer audio Mamba parameters via conditional LoRA to improve multimodal violence detection accuracy.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"b1eae729f6dffafd33b48f1e6f14fe96a3bbc5556838821a6d81f1367134f778"},"source":{"id":"2604.03329","kind":"arxiv","version":2},"verdict":{"id":"7d89e7a9-8aba-4650-b4ee-3fb8d7a901a4","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-13T21:06:14.498550Z","strongest_claim":"On audio-filtered clip-level subsets of NTU-CCTV and DVD, CoLoRSMamba outperforms representative audio-only, video-only, and multimodal baselines, achieving 88.63% accuracy / 86.24% F1-V on NTU-CCTV and 75.77% accuracy / 72.94% F1-V on DVD while offering a favorable accuracy-efficiency tradeoff with fewer parameters and FLOPs.","one_line_summary":"CoLoRSMamba steers AudioMamba using video CLS-guided conditional LoRA to adapt selective state-space parameters, outperforming baselines on audio-filtered NTU-CCTV and DVD subsets with 88.63% and 75.77% accuracy respectively.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the CLS token from VideoMamba can reliably generate channel-wise modulation vectors and stabilization gates that adapt AudioMamba's Delta, B, C parameters to produce genuinely scene-aware audio dynamics without introducing misalignment or noise in real-world conditions.","pith_extraction_headline":"Video CLS tokens steer audio Mamba parameters via conditional LoRA to improve multimodal violence detection accuracy."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2604.03329/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"7d89e7a9-8aba-4650-b4ee-3fb8d7a901a4"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-07T02:18:39Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"pi8bamZmRM9rNkHVpO+GIAr7D3y1Ab/9fMoMZwNQxCB9t+Ki1k4ieJ4pCWeYS2wJiP2O2iConAyEeRFs/f+YBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-20T10:55:59.641698Z"},"content_sha256":"8b2598e79e262c81394a3d39d59c45d3a40ed6400f2a88ee83b9102c99f77381","schema_version":"1.0","event_id":"sha256:8b2598e79e262c81394a3d39d59c45d3a40ed6400f2a88ee83b9102c99f77381"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/FDXUCHFTMJF4H7XRHKYNLNHYA5/bundle.json","state_url":"https://pith.science/pith/FDXUCHFTMJF4H7XRHKYNLNHYA5/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/FDXUCHFTMJF4H7XRHKYNLNHYA5/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-07-20T10:55:59Z","links":{"resolver":"https://pith.science/pith/FDXUCHFTMJF4H7XRHKYNLNHYA5","bundle":"https://pith.science/pith/FDXUCHFTMJF4H7XRHKYNLNHYA5/bundle.json","state":"https://pith.science/pith/FDXUCHFTMJF4H7XRHKYNLNHYA5/state.json","well_known_bundle":"https://pith.science/.well-known/pith/FDXUCHFTMJF4H7XRHKYNLNHYA5/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:FDXUCHFTMJF4H7XRHKYNLNHYA5","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"92c57b3adfce891048610ac17661527dfcd3a71c3cb7e0c30b48c00e3a3b5a25","cross_cats_sorted":["cs.AI","cs.LG","cs.SD"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-04-02T22:14:25Z","title_canon_sha256":"e8cab262b431ef21758c7550d53c6438f5b737844fd2fcdfb66a1b470757e399"},"schema_version":"1.0","source":{"id":"2604.03329","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.03329","created_at":"2026-07-07T02:18:39Z"},{"alias_kind":"arxiv_version","alias_value":"2604.03329v2","created_at":"2026-07-07T02:18:39Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.03329","created_at":"2026-07-07T02:18:39Z"},{"alias_kind":"pith_short_12","alias_value":"FDXUCHFTMJF4","created_at":"2026-07-07T02:18:39Z"},{"alias_kind":"pith_short_16","alias_value":"FDXUCHFTMJF4H7XR","created_at":"2026-07-07T02:18:39Z"},{"alias_kind":"pith_short_8","alias_value":"FDXUCHFT","created_at":"2026-07-07T02:18:39Z"}],"graph_snapshots":[{"event_id":"sha256:8b2598e79e262c81394a3d39d59c45d3a40ed6400f2a88ee83b9102c99f77381","target":"graph","created_at":"2026-07-07T02:18:39Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"On audio-filtered clip-level subsets of NTU-CCTV and DVD, CoLoRSMamba outperforms representative audio-only, video-only, and multimodal baselines, achieving 88.63% accuracy / 86.24% F1-V on NTU-CCTV and 75.77% accuracy / 72.94% F1-V on DVD while offering a favorable accuracy-efficiency tradeoff with fewer parameters and FLOPs."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That the CLS token from VideoMamba can reliably generate channel-wise modulation vectors and stabilization gates that adapt AudioMamba's Delta, B, C parameters to produce genuinely scene-aware audio dynamics without introducing misalignment or noise in real-world conditions."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"CoLoRSMamba steers AudioMamba using video CLS-guided conditional LoRA to adapt selective state-space parameters, outperforming baselines on audio-filtered NTU-CCTV and DVD subsets with 88.63% and 75.77% accuracy respectively."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Video CLS tokens steer audio Mamba parameters via conditional LoRA to improve multimodal violence detection accuracy."}],"snapshot_sha256":"b1eae729f6dffafd33b48f1e6f14fe96a3bbc5556838821a6d81f1367134f778"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2604.03329/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Automatic violence detection from video is challenging because violent interactions may be distant, occluded, or only partially visible. Audio can provide complementary evidence for violent events that are difficult to recognize from visual information alone. However, audio itself may be absent, dubbed, or dominated by environmental noise, making the central challenge not whether to incorporate audio but how to adapt reliance on it according to the visual scene. We introduce \\emph{AViS-Mamba}, an audiovisual Mamba-based architecture in which the visual stream directly governs the behavior of t","authors_text":"Damith Chamalke Senadeera, Dimitrios Kollias, Gregory Slabaugh","cross_cats":["cs.AI","cs.LG","cs.SD"],"headline":"Video CLS tokens steer audio Mamba parameters via conditional LoRA to improve multimodal violence detection accuracy.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-04-02T22:14:25Z","title":"AViS-Mamba: Adaptive Visual Steering of Audio State-Space Dynamics for Violence Detection"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2604.03329","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-13T21:06:14.498550Z","id":"7d89e7a9-8aba-4650-b4ee-3fb8d7a901a4","model_set":{"reader":"grok-4.3"},"one_line_summary":"CoLoRSMamba steers AudioMamba using video CLS-guided conditional LoRA to adapt selective state-space parameters, outperforming baselines on audio-filtered NTU-CCTV and DVD subsets with 88.63% and 75.77% accuracy respectively.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Video CLS tokens steer audio Mamba parameters via conditional LoRA to improve multimodal violence detection accuracy.","strongest_claim":"On audio-filtered clip-level subsets of NTU-CCTV and DVD, CoLoRSMamba outperforms representative audio-only, video-only, and multimodal baselines, achieving 88.63% accuracy / 86.24% F1-V on NTU-CCTV and 75.77% accuracy / 72.94% F1-V on DVD while offering a favorable accuracy-efficiency tradeoff with fewer parameters and FLOPs.","weakest_assumption":"That the CLS token from VideoMamba can reliably generate channel-wise modulation vectors and stabilization gates that adapt AudioMamba's Delta, B, C parameters to produce genuinely scene-aware audio dynamics without introducing misalignment or noise in real-world conditions."}},"verdict_id":"7d89e7a9-8aba-4650-b4ee-3fb8d7a901a4"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:25d797b5db537c095103a99f32e4cb0e57730c5c4bc4cbb574abe7a2a03e48fd","target":"record","created_at":"2026-07-07T02:18:39Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"92c57b3adfce891048610ac17661527dfcd3a71c3cb7e0c30b48c00e3a3b5a25","cross_cats_sorted":["cs.AI","cs.LG","cs.SD"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-04-02T22:14:25Z","title_canon_sha256":"e8cab262b431ef21758c7550d53c6438f5b737844fd2fcdfb66a1b470757e399"},"schema_version":"1.0","source":{"id":"2604.03329","kind":"arxiv","version":2}},"canonical_sha256":"28ef411cb3624bc3fef13ab0d5b4f8074249d28929eb08f580251bd43ce8ed0b","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"28ef411cb3624bc3fef13ab0d5b4f8074249d28929eb08f580251bd43ce8ed0b","first_computed_at":"2026-07-07T02:18:39.570096Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-07T02:18:39.570096Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"3+/5zx+SzvsmbhCpdCfZfuw1bE6FwIvKYlAlB1TeVAS1Wy0tCspXJtfgcpOS70BfKxjrhw739rellC5d2kwMCg==","signature_status":"signed_v1","signed_at":"2026-07-07T02:18:39.571046Z","signed_message":"canonical_sha256_bytes"},"source_id":"2604.03329","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:25d797b5db537c095103a99f32e4cb0e57730c5c4bc4cbb574abe7a2a03e48fd","sha256:8b2598e79e262c81394a3d39d59c45d3a40ed6400f2a88ee83b9102c99f77381"],"state_sha256":"4f244c03f0e0f6807f61732dd33298b61204cb9bda09f42806d17e23777e0a5d"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"a1SqdQb/tAEzgh8Yvmvtr1S5f/VWVk0lzcqX4gIlLlBwId2dryEckmkUFUXvsEr0dl76hkOcyQWwq1MSrBkMDg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-07-20T10:55:59.644022Z","bundle_sha256":"310d07cdbd4bd5bd37497f379901133015a9fa8d2c511ff936a74e7fbf1b3017"}}