{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DJY2XFNW5JMNA23TOKXIMRAYIA","short_pith_number":"pith:DJY2XFNW","schema_version":"1.0","canonical_sha256":"1a71ab95b6ea58d06b7372ae864418403dacd05a20b2a103f06b508e146ad1aa","source":{"kind":"arxiv","id":"2505.17423","version":4},"attestation_state":"computed","paper":{"title":"VIBE: Annotation-Free Video-to-Text Information Bottleneck Evaluation for TL;DR","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.HC","cs.IT","math.IT"],"primary_cat":"cs.CV","authors_text":"Po-han Li, Sandeep Chinchali, Shenghui Chen, Ufuk Topcu","submitted_at":"2025-05-23T03:11:29Z","abstract_excerpt":"Many decision-making tasks, where both accuracy and efficiency matter, still require human supervision. For example, tasks like traffic officers reviewing hour-long dashcam footage or researchers screening conference videos can benefit from concise summaries that reduce cognitive load and save time. Yet current vision-language models (VLMs) often produce verbose, redundant outputs that hinder task performance. Existing video caption evaluation depends on costly human annotations and overlooks the summaries' utility in downstream tasks. We address these gaps with Video-to-text Information Bottl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.17423","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-23T03:11:29Z","cross_cats_sorted":["cs.HC","cs.IT","math.IT"],"title_canon_sha256":"d3d92398819e3299382a491935f0fed099efe7a8b3d5cd356fafc88a3d333f26","abstract_canon_sha256":"669026bb66f35d1ca797ce2b161163c7aab0b331762f4489a4e68a3a2142a30a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-14T02:21:17.451044Z","signature_b64":"IWPKaBm2xJTm6lwr7KCkoE3fjL10iuJjv+BoFFA+OGT3XfLfSiBQgcsmewYZbHI0d7x/cAeRLVJw/iv2uCooDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1a71ab95b6ea58d06b7372ae864418403dacd05a20b2a103f06b508e146ad1aa","last_reissued_at":"2026-07-14T02:21:17.450064Z","signature_status":"signed_v1","first_computed_at":"2026-07-14T02:21:17.450064Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VIBE: Annotation-Free Video-to-Text Information Bottleneck Evaluation for TL;DR","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.HC","cs.IT","math.IT"],"primary_cat":"cs.CV","authors_text":"Po-han Li, Sandeep Chinchali, Shenghui Chen, Ufuk Topcu","submitted_at":"2025-05-23T03:11:29Z","abstract_excerpt":"Many decision-making tasks, where both accuracy and efficiency matter, still require human supervision. For example, tasks like traffic officers reviewing hour-long dashcam footage or researchers screening conference videos can benefit from concise summaries that reduce cognitive load and save time. Yet current vision-language models (VLMs) often produce verbose, redundant outputs that hinder task performance. Existing video caption evaluation depends on costly human annotations and overlooks the summaries' utility in downstream tasks. We address these gaps with Video-to-text Information Bottl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.17423","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.17423/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.17423","created_at":"2026-07-14T02:21:17.450528+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.17423v4","created_at":"2026-07-14T02:21:17.450528+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.17423","created_at":"2026-07-14T02:21:17.450528+00:00"},{"alias_kind":"pith_short_12","alias_value":"DJY2XFNW5JMN","created_at":"2026-07-14T02:21:17.450528+00:00"},{"alias_kind":"pith_short_16","alias_value":"DJY2XFNW5JMNA23T","created_at":"2026-07-14T02:21:17.450528+00:00"},{"alias_kind":"pith_short_8","alias_value":"DJY2XFNW","created_at":"2026-07-14T02:21:17.450528+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DJY2XFNW5JMNA23TOKXIMRAYIA","json":"https://pith.science/pith/DJY2XFNW5JMNA23TOKXIMRAYIA.json","graph_json":"https://pith.science/api/pith-number/DJY2XFNW5JMNA23TOKXIMRAYIA/graph.json","events_json":"https://pith.science/api/pith-number/DJY2XFNW5JMNA23TOKXIMRAYIA/events.json","paper":"https://pith.science/paper/DJY2XFNW"},"agent_actions":{"view_html":"https://pith.science/pith/DJY2XFNW5JMNA23TOKXIMRAYIA","download_json":"https://pith.science/pith/DJY2XFNW5JMNA23TOKXIMRAYIA.json","view_paper":"https://pith.science/paper/DJY2XFNW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.17423&json=true","fetch_graph":"https://pith.science/api/pith-number/DJY2XFNW5JMNA23TOKXIMRAYIA/graph.json","fetch_events":"https://pith.science/api/pith-number/DJY2XFNW5JMNA23TOKXIMRAYIA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DJY2XFNW5JMNA23TOKXIMRAYIA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DJY2XFNW5JMNA23TOKXIMRAYIA/action/storage_attestation","attest_author":"https://pith.science/pith/DJY2XFNW5JMNA23TOKXIMRAYIA/action/author_attestation","sign_citation":"https://pith.science/pith/DJY2XFNW5JMNA23TOKXIMRAYIA/action/citation_signature","submit_replication":"https://pith.science/pith/DJY2XFNW5JMNA23TOKXIMRAYIA/action/replication_record"}},"created_at":"2026-07-14T02:21:17.450528+00:00","updated_at":"2026-07-14T02:21:17.450528+00:00"}