{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GYXLWB4P6ZYM57S2QKNWRZMKMS","short_pith_number":"pith:GYXLWB4P","schema_version":"1.0","canonical_sha256":"362ebb078ff670cefe5a829b68e58a649fa7da24b78a435a18f6df74660352b4","source":{"kind":"arxiv","id":"2412.01095","version":3},"attestation_state":"computed","paper":{"title":"VERA: Explainable Video Anomaly Detection via Verbalized Learning of Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.AI","authors_text":"Muchao Ye, Pan He, Weiyang Liu","submitted_at":"2024-12-02T04:10:14Z","abstract_excerpt":"The rapid advancement of vision-language models (VLMs) has established a new paradigm in video anomaly detection (VAD): leveraging VLMs to simultaneously detect anomalies and provide comprehendible explanations for the decisions. Existing work in this direction often assumes the complex reasoning required for VAD exceeds the capabilities of pretrained VLMs. Consequently, these approaches either incorporate specialized reasoning modules during inference or rely on instruction tuning datasets through additional training to adapt VLMs for VAD. However, such strategies often incur substantial comp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.01095","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-12-02T04:10:14Z","cross_cats_sorted":["cs.CV","cs.LG"],"title_canon_sha256":"d29dc0b869fc65e3f112658c6ebc882bf546b89b214fd98a0708b2bed483079f","abstract_canon_sha256":"b0d81e3217a2a163a138af953ab105b6afb2339810c85ae269428e1ebf42c4a1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:42:14.099421Z","signature_b64":"lFwKKHqmNW/LRVH0RjsUcoDs94FQc7VrRJQ155T4iBaM/mDrqeDbXCAUtq52SdHzQnt6Rm8ITfHMGt3fprdsAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"362ebb078ff670cefe5a829b68e58a649fa7da24b78a435a18f6df74660352b4","last_reissued_at":"2026-07-05T10:42:14.098941Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:42:14.098941Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VERA: Explainable Video Anomaly Detection via Verbalized Learning of Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.AI","authors_text":"Muchao Ye, Pan He, Weiyang Liu","submitted_at":"2024-12-02T04:10:14Z","abstract_excerpt":"The rapid advancement of vision-language models (VLMs) has established a new paradigm in video anomaly detection (VAD): leveraging VLMs to simultaneously detect anomalies and provide comprehendible explanations for the decisions. Existing work in this direction often assumes the complex reasoning required for VAD exceeds the capabilities of pretrained VLMs. Consequently, these approaches either incorporate specialized reasoning modules during inference or rely on instruction tuning datasets through additional training to adapt VLMs for VAD. However, such strategies often incur substantial comp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.01095","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.01095/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.01095","created_at":"2026-07-05T10:42:14.098993+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.01095v3","created_at":"2026-07-05T10:42:14.098993+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.01095","created_at":"2026-07-05T10:42:14.098993+00:00"},{"alias_kind":"pith_short_12","alias_value":"GYXLWB4P6ZYM","created_at":"2026-07-05T10:42:14.098993+00:00"},{"alias_kind":"pith_short_16","alias_value":"GYXLWB4P6ZYM57S2","created_at":"2026-07-05T10:42:14.098993+00:00"},{"alias_kind":"pith_short_8","alias_value":"GYXLWB4P","created_at":"2026-07-05T10:42:14.098993+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.21507","citing_title":"VAGU & GtS: LLM-Based Benchmark and Framework for Joint Video Anomaly Grounding and Understanding","ref_index":17,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GYXLWB4P6ZYM57S2QKNWRZMKMS","json":"https://pith.science/pith/GYXLWB4P6ZYM57S2QKNWRZMKMS.json","graph_json":"https://pith.science/api/pith-number/GYXLWB4P6ZYM57S2QKNWRZMKMS/graph.json","events_json":"https://pith.science/api/pith-number/GYXLWB4P6ZYM57S2QKNWRZMKMS/events.json","paper":"https://pith.science/paper/GYXLWB4P"},"agent_actions":{"view_html":"https://pith.science/pith/GYXLWB4P6ZYM57S2QKNWRZMKMS","download_json":"https://pith.science/pith/GYXLWB4P6ZYM57S2QKNWRZMKMS.json","view_paper":"https://pith.science/paper/GYXLWB4P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.01095&json=true","fetch_graph":"https://pith.science/api/pith-number/GYXLWB4P6ZYM57S2QKNWRZMKMS/graph.json","fetch_events":"https://pith.science/api/pith-number/GYXLWB4P6ZYM57S2QKNWRZMKMS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GYXLWB4P6ZYM57S2QKNWRZMKMS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GYXLWB4P6ZYM57S2QKNWRZMKMS/action/storage_attestation","attest_author":"https://pith.science/pith/GYXLWB4P6ZYM57S2QKNWRZMKMS/action/author_attestation","sign_citation":"https://pith.science/pith/GYXLWB4P6ZYM57S2QKNWRZMKMS/action/citation_signature","submit_replication":"https://pith.science/pith/GYXLWB4P6ZYM57S2QKNWRZMKMS/action/replication_record"}},"created_at":"2026-07-05T10:42:14.098993+00:00","updated_at":"2026-07-05T10:42:14.098993+00:00"}