{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:M7A6QVI5A4JP2K6DHR4XR45CT7","short_pith_number":"pith:M7A6QVI5","schema_version":"1.0","canonical_sha256":"67c1e8551d0712fd2bc33c7978f3a29fc55dd50221331cf4fc6b73425aad9edc","source":{"kind":"arxiv","id":"2407.06189","version":1},"attestation_state":"computed","paper":{"title":"Video-STaR: Self-Training Enables Video Instruction Tuning with Any Supervision","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Idan Szpektor, Orr Zohar, Serena Yeung-Levy, Xiaohan Wang, Yonatan Bitton","submitted_at":"2024-07-08T17:59:42Z","abstract_excerpt":"The performance of Large Vision Language Models (LVLMs) is dependent on the size and quality of their training datasets. Existing video instruction tuning datasets lack diversity as they are derived by prompting large language models with video captions to generate question-answer pairs, and are therefore mostly descriptive. Meanwhile, many labeled video datasets with diverse labels and supervision exist - however, we find that their integration into LVLMs is non-trivial. Herein, we present Video Self-Training with augmented Reasoning (Video-STaR), the first video self-training approach. Video"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.06189","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-07-08T17:59:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"bf40e205e108e7a5289b3e5ac1df2d6ba1bb7c7914557a41c6dab8e25f7154a0","abstract_canon_sha256":"8463f466a5e92b39fa531553ea76092cd0e0ccff336d4279066b1c73875dd94a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:41:24.986447Z","signature_b64":"HR1rohKQ8CPRm3ip/iZUWp/WUkNXUizFooZmzWPpFwUdABnUG1Kqjt7YYEI5pre0i58d3vv7I/k7m57fp5cLBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"67c1e8551d0712fd2bc33c7978f3a29fc55dd50221331cf4fc6b73425aad9edc","last_reissued_at":"2026-07-05T08:41:24.986028Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:41:24.986028Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Video-STaR: Self-Training Enables Video Instruction Tuning with Any Supervision","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Idan Szpektor, Orr Zohar, Serena Yeung-Levy, Xiaohan Wang, Yonatan Bitton","submitted_at":"2024-07-08T17:59:42Z","abstract_excerpt":"The performance of Large Vision Language Models (LVLMs) is dependent on the size and quality of their training datasets. Existing video instruction tuning datasets lack diversity as they are derived by prompting large language models with video captions to generate question-answer pairs, and are therefore mostly descriptive. Meanwhile, many labeled video datasets with diverse labels and supervision exist - however, we find that their integration into LVLMs is non-trivial. Herein, we present Video Self-Training with augmented Reasoning (Video-STaR), the first video self-training approach. Video"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.06189","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.06189/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.06189","created_at":"2026-07-05T08:41:24.986083+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.06189v1","created_at":"2026-07-05T08:41:24.986083+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.06189","created_at":"2026-07-05T08:41:24.986083+00:00"},{"alias_kind":"pith_short_12","alias_value":"M7A6QVI5A4JP","created_at":"2026-07-05T08:41:24.986083+00:00"},{"alias_kind":"pith_short_16","alias_value":"M7A6QVI5A4JP2K6D","created_at":"2026-07-05T08:41:24.986083+00:00"},{"alias_kind":"pith_short_8","alias_value":"M7A6QVI5","created_at":"2026-07-05T08:41:24.986083+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22158","citing_title":"Improving Reasoning in Vision-Language Models via Perception Verified Self-Training","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05259","citing_title":"VideoKR: Towards Knowledge- and Reasoning-Intensive Video Understanding","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22158","citing_title":"Improving Reasoning in Vision-Language Models via Perception Verified Self-Training","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14164","citing_title":"MetaMorph: Multimodal Understanding and Generation via Instruction Tuning","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2504.05299","citing_title":"SmolVLM: Redefining small and efficient multimodal models","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26707","citing_title":"CurEvo: Curriculum-Guided Self-Evolution for Video Understanding","ref_index":102,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M7A6QVI5A4JP2K6DHR4XR45CT7","json":"https://pith.science/pith/M7A6QVI5A4JP2K6DHR4XR45CT7.json","graph_json":"https://pith.science/api/pith-number/M7A6QVI5A4JP2K6DHR4XR45CT7/graph.json","events_json":"https://pith.science/api/pith-number/M7A6QVI5A4JP2K6DHR4XR45CT7/events.json","paper":"https://pith.science/paper/M7A6QVI5"},"agent_actions":{"view_html":"https://pith.science/pith/M7A6QVI5A4JP2K6DHR4XR45CT7","download_json":"https://pith.science/pith/M7A6QVI5A4JP2K6DHR4XR45CT7.json","view_paper":"https://pith.science/paper/M7A6QVI5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.06189&json=true","fetch_graph":"https://pith.science/api/pith-number/M7A6QVI5A4JP2K6DHR4XR45CT7/graph.json","fetch_events":"https://pith.science/api/pith-number/M7A6QVI5A4JP2K6DHR4XR45CT7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M7A6QVI5A4JP2K6DHR4XR45CT7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M7A6QVI5A4JP2K6DHR4XR45CT7/action/storage_attestation","attest_author":"https://pith.science/pith/M7A6QVI5A4JP2K6DHR4XR45CT7/action/author_attestation","sign_citation":"https://pith.science/pith/M7A6QVI5A4JP2K6DHR4XR45CT7/action/citation_signature","submit_replication":"https://pith.science/pith/M7A6QVI5A4JP2K6DHR4XR45CT7/action/replication_record"}},"created_at":"2026-07-05T08:41:24.986083+00:00","updated_at":"2026-07-05T08:41:24.986083+00:00"}