{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2NRXITEW2VAAFD7X3FWYW2FHS4","short_pith_number":"pith:2NRXITEW","schema_version":"1.0","canonical_sha256":"d363744c96d540028ff7d96d8b68a797074265905ad4ce65588f33c6126214f8","source":{"kind":"arxiv","id":"2506.07971","version":1},"attestation_state":"computed","paper":{"title":"CyberV: Cybernetics for Test-time Scaling in Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiahao Meng, Longyin Wen, Lu Qi, Shuyang Sun, Xiangtai Li, Yue Tan, Yunhai Tong","submitted_at":"2025-06-09T17:45:18Z","abstract_excerpt":"Current Multimodal Large Language Models (MLLMs) may struggle with understanding long or complex videos due to computational demands at test time, lack of robustness, and limited accuracy, primarily stemming from their feed-forward processing nature. These limitations could be more severe for models with fewer parameters. To address these limitations, we propose a novel framework inspired by cybernetic principles, redesigning video MLLMs as adaptive systems capable of self-monitoring, self-correction, and dynamic resource allocation during inference. Our approach, CyberV, introduces a cybernet"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.07971","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-06-09T17:45:18Z","cross_cats_sorted":[],"title_canon_sha256":"c97e2bcd8eae285d90b2006b6f6d7296a2ae60568490dd34439098fd42857496","abstract_canon_sha256":"a63c760bc3574cb615a87e0b5fa1451784ae4c022528eeb6d7c86c6ade1a69ce"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:36.426456Z","signature_b64":"pL1e0iSacXfV/Zad5dcuI56L2XhmED2r80mRIpx2qHE69cn/AQ2OQLgaTaVUzPAnIYwriwsFCaQW8t8uwPF7DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d363744c96d540028ff7d96d8b68a797074265905ad4ce65588f33c6126214f8","last_reissued_at":"2026-07-05T11:18:36.425894Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:36.425894Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CyberV: Cybernetics for Test-time Scaling in Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiahao Meng, Longyin Wen, Lu Qi, Shuyang Sun, Xiangtai Li, Yue Tan, Yunhai Tong","submitted_at":"2025-06-09T17:45:18Z","abstract_excerpt":"Current Multimodal Large Language Models (MLLMs) may struggle with understanding long or complex videos due to computational demands at test time, lack of robustness, and limited accuracy, primarily stemming from their feed-forward processing nature. These limitations could be more severe for models with fewer parameters. To address these limitations, we propose a novel framework inspired by cybernetic principles, redesigning video MLLMs as adaptive systems capable of self-monitoring, self-correction, and dynamic resource allocation during inference. Our approach, CyberV, introduces a cybernet"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.07971","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.07971/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.07971","created_at":"2026-07-05T11:18:36.425960+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.07971v1","created_at":"2026-07-05T11:18:36.425960+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.07971","created_at":"2026-07-05T11:18:36.425960+00:00"},{"alias_kind":"pith_short_12","alias_value":"2NRXITEW2VAA","created_at":"2026-07-05T11:18:36.425960+00:00"},{"alias_kind":"pith_short_16","alias_value":"2NRXITEW2VAAFD7X","created_at":"2026-07-05T11:18:36.425960+00:00"},{"alias_kind":"pith_short_8","alias_value":"2NRXITEW","created_at":"2026-07-05T11:18:36.425960+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11576","citing_title":"AVIS: Adaptive Test-Time Scaling for Vision-Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08231","citing_title":"Test-Time Scaling in Multimodal Foundation Models: A Comprehensive Survey of Generation and Reasoning","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07433","citing_title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","ref_index":229,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06294","citing_title":"Towards One-to-Many Temporal Grounding","ref_index":51,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2NRXITEW2VAAFD7X3FWYW2FHS4","json":"https://pith.science/pith/2NRXITEW2VAAFD7X3FWYW2FHS4.json","graph_json":"https://pith.science/api/pith-number/2NRXITEW2VAAFD7X3FWYW2FHS4/graph.json","events_json":"https://pith.science/api/pith-number/2NRXITEW2VAAFD7X3FWYW2FHS4/events.json","paper":"https://pith.science/paper/2NRXITEW"},"agent_actions":{"view_html":"https://pith.science/pith/2NRXITEW2VAAFD7X3FWYW2FHS4","download_json":"https://pith.science/pith/2NRXITEW2VAAFD7X3FWYW2FHS4.json","view_paper":"https://pith.science/paper/2NRXITEW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.07971&json=true","fetch_graph":"https://pith.science/api/pith-number/2NRXITEW2VAAFD7X3FWYW2FHS4/graph.json","fetch_events":"https://pith.science/api/pith-number/2NRXITEW2VAAFD7X3FWYW2FHS4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2NRXITEW2VAAFD7X3FWYW2FHS4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2NRXITEW2VAAFD7X3FWYW2FHS4/action/storage_attestation","attest_author":"https://pith.science/pith/2NRXITEW2VAAFD7X3FWYW2FHS4/action/author_attestation","sign_citation":"https://pith.science/pith/2NRXITEW2VAAFD7X3FWYW2FHS4/action/citation_signature","submit_replication":"https://pith.science/pith/2NRXITEW2VAAFD7X3FWYW2FHS4/action/replication_record"}},"created_at":"2026-07-05T11:18:36.425960+00:00","updated_at":"2026-07-05T11:18:36.425960+00:00"}