{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CVFQIGAW74JQBETROGSOBSCMA5","short_pith_number":"pith:CVFQIGAW","schema_version":"1.0","canonical_sha256":"154b041816ff1300927171a4e0c84c076f4a007cf33bfa981f73528036671039","source":{"kind":"arxiv","id":"2401.05702","version":1},"attestation_state":"computed","paper":{"title":"Video Anomaly Detection and Explanation via Large Language Models","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hui Lv, Qianru Sun","submitted_at":"2024-01-11T07:09:44Z","abstract_excerpt":"Video Anomaly Detection (VAD) aims to localize abnormal events on the timeline of long-range surveillance videos. Anomaly-scoring-based methods have been prevailing for years but suffer from the high complexity of thresholding and low explanability of detection results. In this paper, we conduct pioneer research on equipping video-based large language models (VLLMs) in the framework of VAD, making the VAD model free from thresholds and able to explain the reasons for the detected anomalies. We introduce a novel network module Long-Term Context (LTC) to mitigate the incapability of VLLMs in lon"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.05702","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CV","submitted_at":"2024-01-11T07:09:44Z","cross_cats_sorted":[],"title_canon_sha256":"5c0de076688c0ff97194abb5479684ab1abffc0b5070407e9db987bec523adab","abstract_canon_sha256":"ef211d14ec453d470683d541547f7bd0732bceab643244cbbd638673f05aaae0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:32:34.171670Z","signature_b64":"lSmQ5JOXKVu/u7KzI6qzK6YIcG+Pd1Izm3CVT8pdREyIzlni5zu+/KkZ4U7KR49CvAkMYAQpYeDBFa726mfLCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"154b041816ff1300927171a4e0c84c076f4a007cf33bfa981f73528036671039","last_reissued_at":"2026-07-05T07:32:34.171137Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:32:34.171137Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Video Anomaly Detection and Explanation via Large Language Models","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hui Lv, Qianru Sun","submitted_at":"2024-01-11T07:09:44Z","abstract_excerpt":"Video Anomaly Detection (VAD) aims to localize abnormal events on the timeline of long-range surveillance videos. Anomaly-scoring-based methods have been prevailing for years but suffer from the high complexity of thresholding and low explanability of detection results. In this paper, we conduct pioneer research on equipping video-based large language models (VLLMs) in the framework of VAD, making the VAD model free from thresholds and able to explain the reasons for the detected anomalies. We introduce a novel network module Long-Term Context (LTC) to mitigate the incapability of VLLMs in lon"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.05702","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.05702/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.05702","created_at":"2026-07-05T07:32:34.171201+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.05702v1","created_at":"2026-07-05T07:32:34.171201+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.05702","created_at":"2026-07-05T07:32:34.171201+00:00"},{"alias_kind":"pith_short_12","alias_value":"CVFQIGAW74JQ","created_at":"2026-07-05T07:32:34.171201+00:00"},{"alias_kind":"pith_short_16","alias_value":"CVFQIGAW74JQBETR","created_at":"2026-07-05T07:32:34.171201+00:00"},{"alias_kind":"pith_short_8","alias_value":"CVFQIGAW","created_at":"2026-07-05T07:32:34.171201+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.15054","citing_title":"LATERN: Test-Time Context-Aware Explainable Video Anomaly Detection","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23116","citing_title":"CoReVAD: A Contextual Reasoning Framework for Training-Free Video Anomaly Detection","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CVFQIGAW74JQBETROGSOBSCMA5","json":"https://pith.science/pith/CVFQIGAW74JQBETROGSOBSCMA5.json","graph_json":"https://pith.science/api/pith-number/CVFQIGAW74JQBETROGSOBSCMA5/graph.json","events_json":"https://pith.science/api/pith-number/CVFQIGAW74JQBETROGSOBSCMA5/events.json","paper":"https://pith.science/paper/CVFQIGAW"},"agent_actions":{"view_html":"https://pith.science/pith/CVFQIGAW74JQBETROGSOBSCMA5","download_json":"https://pith.science/pith/CVFQIGAW74JQBETROGSOBSCMA5.json","view_paper":"https://pith.science/paper/CVFQIGAW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.05702&json=true","fetch_graph":"https://pith.science/api/pith-number/CVFQIGAW74JQBETROGSOBSCMA5/graph.json","fetch_events":"https://pith.science/api/pith-number/CVFQIGAW74JQBETROGSOBSCMA5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CVFQIGAW74JQBETROGSOBSCMA5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CVFQIGAW74JQBETROGSOBSCMA5/action/storage_attestation","attest_author":"https://pith.science/pith/CVFQIGAW74JQBETROGSOBSCMA5/action/author_attestation","sign_citation":"https://pith.science/pith/CVFQIGAW74JQBETROGSOBSCMA5/action/citation_signature","submit_replication":"https://pith.science/pith/CVFQIGAW74JQBETROGSOBSCMA5/action/replication_record"}},"created_at":"2026-07-05T07:32:34.171201+00:00","updated_at":"2026-07-05T07:32:34.171201+00:00"}