{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YG4YCSMLGGFA74PJCRNCSEZUSY","short_pith_number":"pith:YG4YCSML","schema_version":"1.0","canonical_sha256":"c1b981498b318a0ff1e9145a2913349602da7735584957100232371e01648caa","source":{"kind":"arxiv","id":"2506.07371","version":2},"attestation_state":"computed","paper":{"title":"ARGUS: Hallucination and Omission Evaluation in Video-LLMs","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Gowthami Somepalli, Heng Huang, Reza Shirkavand, Ruchit Rawal, Tom Goldstein","submitted_at":"2025-06-09T02:42:13Z","abstract_excerpt":"Video large language models have not yet been widely deployed, largely due to their tendency to hallucinate. Typical benchmarks for Video-LLMs rely simply on multiple-choice questions. Unfortunately, VideoLLMs hallucinate far more aggressively on freeform text generation tasks like video captioning than they do on multiple choice verification tasks. To address this weakness, we propose ARGUS, a VideoLLM benchmark that measures freeform video captioning performance. By comparing VideoLLM outputs to human ground truth captions, ARGUS quantifies dual metrics. First, we measure the rate of halluci"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.07371","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-06-09T02:42:13Z","cross_cats_sorted":[],"title_canon_sha256":"f36fd8ef3d93f983abfd1b06ada084ef7b640ab9b3a343cf53332dc63be69a27","abstract_canon_sha256":"277f09833b76345c49af1fe0cdd36cc8fac0b06ac956b9c51e2724314f1bd114"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:19:08.302412Z","signature_b64":"xF08RtfDy1l5EkHhmZ0xIPNayNoakgJHcL0MsOQ7qMLnIBxgAkJPc3wYvkn7X1tGnGUYpi0AkUcjzrLAW/BEBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c1b981498b318a0ff1e9145a2913349602da7735584957100232371e01648caa","last_reissued_at":"2026-07-05T11:19:08.301916Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:19:08.301916Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ARGUS: Hallucination and Omission Evaluation in Video-LLMs","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Gowthami Somepalli, Heng Huang, Reza Shirkavand, Ruchit Rawal, Tom Goldstein","submitted_at":"2025-06-09T02:42:13Z","abstract_excerpt":"Video large language models have not yet been widely deployed, largely due to their tendency to hallucinate. Typical benchmarks for Video-LLMs rely simply on multiple-choice questions. Unfortunately, VideoLLMs hallucinate far more aggressively on freeform text generation tasks like video captioning than they do on multiple choice verification tasks. To address this weakness, we propose ARGUS, a VideoLLM benchmark that measures freeform video captioning performance. By comparing VideoLLM outputs to human ground truth captions, ARGUS quantifies dual metrics. First, we measure the rate of halluci"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.07371","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.07371/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.07371","created_at":"2026-07-05T11:19:08.301970+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.07371v2","created_at":"2026-07-05T11:19:08.301970+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.07371","created_at":"2026-07-05T11:19:08.301970+00:00"},{"alias_kind":"pith_short_12","alias_value":"YG4YCSMLGGFA","created_at":"2026-07-05T11:19:08.301970+00:00"},{"alias_kind":"pith_short_16","alias_value":"YG4YCSMLGGFA74PJ","created_at":"2026-07-05T11:19:08.301970+00:00"},{"alias_kind":"pith_short_8","alias_value":"YG4YCSML","created_at":"2026-07-05T11:19:08.301970+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.28023","citing_title":"VCap: Hypergeometric Rewards for Weak-to-Strong Visual Captioning","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22226","citing_title":"Towards Temporal Compositional Reasoning in Long-Form Sports Videos","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YG4YCSMLGGFA74PJCRNCSEZUSY","json":"https://pith.science/pith/YG4YCSMLGGFA74PJCRNCSEZUSY.json","graph_json":"https://pith.science/api/pith-number/YG4YCSMLGGFA74PJCRNCSEZUSY/graph.json","events_json":"https://pith.science/api/pith-number/YG4YCSMLGGFA74PJCRNCSEZUSY/events.json","paper":"https://pith.science/paper/YG4YCSML"},"agent_actions":{"view_html":"https://pith.science/pith/YG4YCSMLGGFA74PJCRNCSEZUSY","download_json":"https://pith.science/pith/YG4YCSMLGGFA74PJCRNCSEZUSY.json","view_paper":"https://pith.science/paper/YG4YCSML","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.07371&json=true","fetch_graph":"https://pith.science/api/pith-number/YG4YCSMLGGFA74PJCRNCSEZUSY/graph.json","fetch_events":"https://pith.science/api/pith-number/YG4YCSMLGGFA74PJCRNCSEZUSY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YG4YCSMLGGFA74PJCRNCSEZUSY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YG4YCSMLGGFA74PJCRNCSEZUSY/action/storage_attestation","attest_author":"https://pith.science/pith/YG4YCSMLGGFA74PJCRNCSEZUSY/action/author_attestation","sign_citation":"https://pith.science/pith/YG4YCSMLGGFA74PJCRNCSEZUSY/action/citation_signature","submit_replication":"https://pith.science/pith/YG4YCSMLGGFA74PJCRNCSEZUSY/action/replication_record"}},"created_at":"2026-07-05T11:19:08.301970+00:00","updated_at":"2026-07-05T11:19:08.301970+00:00"}