{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MXM3LZMEVEK5RG5LP3MHNHYLBT","short_pith_number":"pith:MXM3LZME","schema_version":"1.0","canonical_sha256":"65d9b5e584a915d89bab7ed8769f0b0cf72205424fef7433c4293d0dae87240c","source":{"kind":"arxiv","id":"2505.12589","version":1},"attestation_state":"computed","paper":{"title":"SurveillanceVQA-589K: A Benchmark for Comprehensive Surveillance Video-Language Understanding with Large Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo Liu, Kun Liu, Minhan Ma, Pengfei Qiao, Peng Xu, Tongtong Yuan, Xuange Zhang, Yinan Tang","submitted_at":"2025-05-19T00:57:04Z","abstract_excerpt":"Understanding surveillance video content remains a critical yet underexplored challenge in vision-language research, particularly due to its real-world complexity, irregular event dynamics, and safety-critical implications. In this work, we introduce SurveillanceVQA-589K, the largest open-ended video question answering benchmark tailored to the surveillance domain. The dataset comprises 589,380 QA pairs spanning 12 cognitively diverse question types, including temporal reasoning, causal inference, spatial understanding, and anomaly interpretation, across both normal and abnormal video scenario"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.12589","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-19T00:57:04Z","cross_cats_sorted":[],"title_canon_sha256":"ad21468b53f12e71ff357f596bc120ee4be2d471165a89881b0bb07d0841ec38","abstract_canon_sha256":"eb4c79c7943cf2634e1399ea6aa6237380547ea4ce7615633e2ce0bca93966d0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:05:06.566578Z","signature_b64":"w/mLWaaLWeNn9P5ukvDOhY6nBl80vf8Y87R1SI4o93DwaVsKA61PCRqOxZFIVV2P9SpLOEczCovHWC/xfHZfCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"65d9b5e584a915d89bab7ed8769f0b0cf72205424fef7433c4293d0dae87240c","last_reissued_at":"2026-07-05T11:05:06.566095Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:05:06.566095Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SurveillanceVQA-589K: A Benchmark for Comprehensive Surveillance Video-Language Understanding with Large Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo Liu, Kun Liu, Minhan Ma, Pengfei Qiao, Peng Xu, Tongtong Yuan, Xuange Zhang, Yinan Tang","submitted_at":"2025-05-19T00:57:04Z","abstract_excerpt":"Understanding surveillance video content remains a critical yet underexplored challenge in vision-language research, particularly due to its real-world complexity, irregular event dynamics, and safety-critical implications. In this work, we introduce SurveillanceVQA-589K, the largest open-ended video question answering benchmark tailored to the surveillance domain. The dataset comprises 589,380 QA pairs spanning 12 cognitively diverse question types, including temporal reasoning, causal inference, spatial understanding, and anomaly interpretation, across both normal and abnormal video scenario"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.12589","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.12589/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.12589","created_at":"2026-07-05T11:05:06.566157+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.12589v1","created_at":"2026-07-05T11:05:06.566157+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.12589","created_at":"2026-07-05T11:05:06.566157+00:00"},{"alias_kind":"pith_short_12","alias_value":"MXM3LZMEVEK5","created_at":"2026-07-05T11:05:06.566157+00:00"},{"alias_kind":"pith_short_16","alias_value":"MXM3LZMEVEK5RG5L","created_at":"2026-07-05T11:05:06.566157+00:00"},{"alias_kind":"pith_short_8","alias_value":"MXM3LZME","created_at":"2026-07-05T11:05:06.566157+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.25461","citing_title":"MetaphorVU: Towards Metaphorical Video Understanding","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21917","citing_title":"MAVEN: A Multi-stage Agentic Annotation Pipeline for Video Reasoning Tasks","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2512.03479","citing_title":"ProcObject-10K: Benchmarking Object-Centric Procedural Understanding in Instructional Videos","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08457","citing_title":"CrashSight: A Phase-Aware, Infrastructure-Centric Video Benchmark for Traffic Crash Scene Understanding and Reasoning","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MXM3LZMEVEK5RG5LP3MHNHYLBT","json":"https://pith.science/pith/MXM3LZMEVEK5RG5LP3MHNHYLBT.json","graph_json":"https://pith.science/api/pith-number/MXM3LZMEVEK5RG5LP3MHNHYLBT/graph.json","events_json":"https://pith.science/api/pith-number/MXM3LZMEVEK5RG5LP3MHNHYLBT/events.json","paper":"https://pith.science/paper/MXM3LZME"},"agent_actions":{"view_html":"https://pith.science/pith/MXM3LZMEVEK5RG5LP3MHNHYLBT","download_json":"https://pith.science/pith/MXM3LZMEVEK5RG5LP3MHNHYLBT.json","view_paper":"https://pith.science/paper/MXM3LZME","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.12589&json=true","fetch_graph":"https://pith.science/api/pith-number/MXM3LZMEVEK5RG5LP3MHNHYLBT/graph.json","fetch_events":"https://pith.science/api/pith-number/MXM3LZMEVEK5RG5LP3MHNHYLBT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MXM3LZMEVEK5RG5LP3MHNHYLBT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MXM3LZMEVEK5RG5LP3MHNHYLBT/action/storage_attestation","attest_author":"https://pith.science/pith/MXM3LZMEVEK5RG5LP3MHNHYLBT/action/author_attestation","sign_citation":"https://pith.science/pith/MXM3LZMEVEK5RG5LP3MHNHYLBT/action/citation_signature","submit_replication":"https://pith.science/pith/MXM3LZMEVEK5RG5LP3MHNHYLBT/action/replication_record"}},"created_at":"2026-07-05T11:05:06.566157+00:00","updated_at":"2026-07-05T11:05:06.566157+00:00"}