{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:C3I3AYR4M575W66FKW443Y7HNE","short_pith_number":"pith:C3I3AYR4","canonical_record":{"source":{"id":"2410.02713","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-03T17:36:49Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"92a0fb1ef8367dd6301982756cdb76c64ec78fa0ac1f18ccecf0dad9fe10fcf3","abstract_canon_sha256":"4ede5d7cd55f57a9f6140f02d1e6fdbb7f223ea849ce038d85968e269977d0fd"},"schema_version":"1.0"},"canonical_sha256":"16d1b0623c677fdb7bc555b9cde3e7691cb7601da3c526f43fb022dbdacf3873","source":{"kind":"arxiv","id":"2410.02713","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.02713","created_at":"2026-07-05T11:46:36Z"},{"alias_kind":"arxiv_version","alias_value":"2410.02713v3","created_at":"2026-07-05T11:46:36Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.02713","created_at":"2026-07-05T11:46:36Z"},{"alias_kind":"pith_short_12","alias_value":"C3I3AYR4M575","created_at":"2026-07-05T11:46:36Z"},{"alias_kind":"pith_short_16","alias_value":"C3I3AYR4M575W66F","created_at":"2026-07-05T11:46:36Z"},{"alias_kind":"pith_short_8","alias_value":"C3I3AYR4","created_at":"2026-07-05T11:46:36Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:C3I3AYR4M575W66FKW443Y7HNE","target":"record","payload":{"canonical_record":{"source":{"id":"2410.02713","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-03T17:36:49Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"92a0fb1ef8367dd6301982756cdb76c64ec78fa0ac1f18ccecf0dad9fe10fcf3","abstract_canon_sha256":"4ede5d7cd55f57a9f6140f02d1e6fdbb7f223ea849ce038d85968e269977d0fd"},"schema_version":"1.0"},"canonical_sha256":"16d1b0623c677fdb7bc555b9cde3e7691cb7601da3c526f43fb022dbdacf3873","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:46:36.419714Z","signature_b64":"a4KMgFyB8v4VqNpsLyPy0ZN/tWvZg1Yd77wcGfo0v0+r94kmd5joeuSQRYZrKMFNKMxJ9GQaDRukv1CkdOb7DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"16d1b0623c677fdb7bc555b9cde3e7691cb7601da3c526f43fb022dbdacf3873","last_reissued_at":"2026-07-05T11:46:36.419099Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:46:36.419099Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2410.02713","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:46:36Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"BzM9aZrxXOtvsTc6Eg8f9C4DPbmP+yjz7pUo1xjPfZYqk6DYiyUj9ArFZ3CAvdkLwx8wmzo+HLZi8FgNGux3AQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T00:06:15.836816Z"},"content_sha256":"d459bee5cd3039247e96a6577e0bff5e9e17ed15087e9874116be134f719a813","schema_version":"1.0","event_id":"sha256:d459bee5cd3039247e96a6577e0bff5e9e17ed15087e9874116be134f719a813"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:C3I3AYR4M575W66FKW443Y7HNE","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Synthetic video instructions train a model that performs strongly on real benchmarks.","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Bo Li, Chunyuan Li, Jinming Wu, Wei Li, Yuanhan Zhang, Zejun Ma, Ziwei Liu","submitted_at":"2024-10-03T17:36:49Z","abstract_excerpt":"The development of video large multimodal models (LMMs) has been hindered by the difficulty of curating large amounts of high-quality raw data from the web. To address this, we propose an alternative approach by creating a high-quality synthetic dataset specifically for video instruction-following, namely LLaVA-Video-178K. This dataset includes key tasks such as detailed captioning, open-ended question-answering (QA), and multiple-choice QA. By training on this dataset, in combination with existing visual instruction tuning data, we introduce LLaVA-Video, a new video LMM. Our experiments demon"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"By training on this dataset, in combination with existing visual instruction tuning data, we introduce LLaVA-Video, a new video LMM. Our experiments demonstrate that LLaVA-Video achieves strong performance across various video benchmarks, highlighting the effectiveness of our dataset.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the synthetically generated video instructions are of high enough quality and diversity to produce genuine generalization on real-world video tasks rather than merely fitting the chosen benchmarks.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"LLaVA-Video-178K is a new synthetic video instruction dataset that, when combined with existing data to train LLaVA-Video, produces strong results on video understanding benchmarks.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Synthetic video instructions train a model that performs strongly on real benchmarks.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"70d3f46a842b8317820aec0a7fbb9b18ea0c66062331a87c8ef65e30c7d1eb6b"},"source":{"id":"2410.02713","kind":"arxiv","version":3},"verdict":{"id":"e7019bd8-34fd-4fe5-b441-40d3e598f020","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-10T23:16:19.127751Z","strongest_claim":"By training on this dataset, in combination with existing visual instruction tuning data, we introduce LLaVA-Video, a new video LMM. Our experiments demonstrate that LLaVA-Video achieves strong performance across various video benchmarks, highlighting the effectiveness of our dataset.","one_line_summary":"LLaVA-Video-178K is a new synthetic video instruction dataset that, when combined with existing data to train LLaVA-Video, produces strong results on video understanding benchmarks.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the synthetically generated video instructions are of high enough quality and diversity to produce genuine generalization on real-world video tasks rather than merely fitting the chosen benchmarks.","pith_extraction_headline":"Synthetic video instructions train a model that performs strongly on real benchmarks."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.02713/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":190,"sample":[{"doi":"","year":2022,"title":"Flamingo: a Visual Language Model for Few-Shot Learning","work_id":"a110f764-38dc-41b2-a802-53744ecea1fc","ref_index":1,"cited_arxiv_id":"2204.14198","is_internal_anchor":true},{"doi":"","year":2017,"title":"Localizing moments in video with natural language","work_id":"54060e06-71cc-40ad-9c5b-2f3adf9d88ce","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2017,"title":"Localizing moments in video with natural language","work_id":"d615ddc8-0f7c-4e92-9d99-5f64b773b840","ref_index":3,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2021,"title":"Frozen in time: A joint video and image encoder for end-to-end retrieval","work_id":"d24a093e-c939-4159-86ac-e79139e01093","ref_index":4,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2015,"title":"Activitynet: A large-scale video benchmark for human activity understanding","work_id":"da056c16-524b-48ee-8932-184520fa61cc","ref_index":5,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":190,"snapshot_sha256":"34a40a0578425333315e9d0cebff35ae89cf380dd96692283cd74d0cea42e3a7","internal_anchors":30},"formal_canon":{"evidence_count":3,"snapshot_sha256":"c06cc9423fd4b3479a577637110b5d6f90821a03e528c534dbdb00a83691ca9c"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"e7019bd8-34fd-4fe5-b441-40d3e598f020"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:46:36Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"qT5xS/a3kzyPFZ8XVVZC9cuCupVVhS4vKrwfJTZEB/G0kUfxrAaksohDZupIpWqMIyuCBX4HQDVvj3x9kl+sCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T00:06:15.837980Z"},"content_sha256":"a0cef3bb89205e6588729087b27d96efb9a48460a41a19bd96d34d8feffcac31","schema_version":"1.0","event_id":"sha256:a0cef3bb89205e6588729087b27d96efb9a48460a41a19bd96d34d8feffcac31"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/C3I3AYR4M575W66FKW443Y7HNE/bundle.json","state_url":"https://pith.science/pith/C3I3AYR4M575W66FKW443Y7HNE/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/C3I3AYR4M575W66FKW443Y7HNE/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-11T00:06:15Z","links":{"resolver":"https://pith.science/pith/C3I3AYR4M575W66FKW443Y7HNE","bundle":"https://pith.science/pith/C3I3AYR4M575W66FKW443Y7HNE/bundle.json","state":"https://pith.science/pith/C3I3AYR4M575W66FKW443Y7HNE/state.json","well_known_bundle":"https://pith.science/.well-known/pith/C3I3AYR4M575W66FKW443Y7HNE/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:C3I3AYR4M575W66FKW443Y7HNE","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"4ede5d7cd55f57a9f6140f02d1e6fdbb7f223ea849ce038d85968e269977d0fd","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-03T17:36:49Z","title_canon_sha256":"92a0fb1ef8367dd6301982756cdb76c64ec78fa0ac1f18ccecf0dad9fe10fcf3"},"schema_version":"1.0","source":{"id":"2410.02713","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.02713","created_at":"2026-07-05T11:46:36Z"},{"alias_kind":"arxiv_version","alias_value":"2410.02713v3","created_at":"2026-07-05T11:46:36Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.02713","created_at":"2026-07-05T11:46:36Z"},{"alias_kind":"pith_short_12","alias_value":"C3I3AYR4M575","created_at":"2026-07-05T11:46:36Z"},{"alias_kind":"pith_short_16","alias_value":"C3I3AYR4M575W66F","created_at":"2026-07-05T11:46:36Z"},{"alias_kind":"pith_short_8","alias_value":"C3I3AYR4","created_at":"2026-07-05T11:46:36Z"}],"graph_snapshots":[{"event_id":"sha256:a0cef3bb89205e6588729087b27d96efb9a48460a41a19bd96d34d8feffcac31","target":"graph","created_at":"2026-07-05T11:46:36Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"By training on this dataset, in combination with existing visual instruction tuning data, we introduce LLaVA-Video, a new video LMM. Our experiments demonstrate that LLaVA-Video achieves strong performance across various video benchmarks, highlighting the effectiveness of our dataset."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That the synthetically generated video instructions are of high enough quality and diversity to produce genuine generalization on real-world video tasks rather than merely fitting the chosen benchmarks."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"LLaVA-Video-178K is a new synthetic video instruction dataset that, when combined with existing data to train LLaVA-Video, produces strong results on video understanding benchmarks."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Synthetic video instructions train a model that performs strongly on real benchmarks."}],"snapshot_sha256":"70d3f46a842b8317820aec0a7fbb9b18ea0c66062331a87c8ef65e30c7d1eb6b"},"formal_canon":{"evidence_count":3,"snapshot_sha256":"c06cc9423fd4b3479a577637110b5d6f90821a03e528c534dbdb00a83691ca9c"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2410.02713/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The development of video large multimodal models (LMMs) has been hindered by the difficulty of curating large amounts of high-quality raw data from the web. To address this, we propose an alternative approach by creating a high-quality synthetic dataset specifically for video instruction-following, namely LLaVA-Video-178K. This dataset includes key tasks such as detailed captioning, open-ended question-answering (QA), and multiple-choice QA. By training on this dataset, in combination with existing visual instruction tuning data, we introduce LLaVA-Video, a new video LMM. Our experiments demon","authors_text":"Bo Li, Chunyuan Li, Jinming Wu, Wei Li, Yuanhan Zhang, Zejun Ma, Ziwei Liu","cross_cats":["cs.CL"],"headline":"Synthetic video instructions train a model that performs strongly on real benchmarks.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data"},"references":{"count":190,"internal_anchors":30,"resolved_work":190,"sample":[{"cited_arxiv_id":"2204.14198","doi":"","is_internal_anchor":true,"ref_index":1,"title":"Flamingo: a Visual Language Model for Few-Shot Learning","work_id":"a110f764-38dc-41b2-a802-53744ecea1fc","year":2022},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":2,"title":"Localizing moments in video with natural language","work_id":"54060e06-71cc-40ad-9c5b-2f3adf9d88ce","year":2017},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":3,"title":"Localizing moments in video with natural language","work_id":"d615ddc8-0f7c-4e92-9d99-5f64b773b840","year":2017},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":4,"title":"Frozen in time: A joint video and image encoder for end-to-end retrieval","work_id":"d24a093e-c939-4159-86ac-e79139e01093","year":2021},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":5,"title":"Activitynet: A large-scale video benchmark for human activity understanding","work_id":"da056c16-524b-48ee-8932-184520fa61cc","year":2015}],"snapshot_sha256":"34a40a0578425333315e9d0cebff35ae89cf380dd96692283cd74d0cea42e3a7"},"source":{"id":"2410.02713","kind":"arxiv","version":3},"verdict":{"created_at":"2026-05-10T23:16:19.127751Z","id":"e7019bd8-34fd-4fe5-b441-40d3e598f020","model_set":{"reader":"grok-4.3"},"one_line_summary":"LLaVA-Video-178K is a new synthetic video instruction dataset that, when combined with existing data to train LLaVA-Video, produces strong results on video understanding benchmarks.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Synthetic video instructions train a model that performs strongly on real benchmarks.","strongest_claim":"By training on this dataset, in combination with existing visual instruction tuning data, we introduce LLaVA-Video, a new video LMM. Our experiments demonstrate that LLaVA-Video achieves strong performance across various video benchmarks, highlighting the effectiveness of our dataset.","weakest_assumption":"That the synthetically generated video instructions are of high enough quality and diversity to produce genuine generalization on real-world video tasks rather than merely fitting the chosen benchmarks."}},"verdict_id":"e7019bd8-34fd-4fe5-b441-40d3e598f020"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:d459bee5cd3039247e96a6577e0bff5e9e17ed15087e9874116be134f719a813","target":"record","created_at":"2026-07-05T11:46:36Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"4ede5d7cd55f57a9f6140f02d1e6fdbb7f223ea849ce038d85968e269977d0fd","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-03T17:36:49Z","title_canon_sha256":"92a0fb1ef8367dd6301982756cdb76c64ec78fa0ac1f18ccecf0dad9fe10fcf3"},"schema_version":"1.0","source":{"id":"2410.02713","kind":"arxiv","version":3}},"canonical_sha256":"16d1b0623c677fdb7bc555b9cde3e7691cb7601da3c526f43fb022dbdacf3873","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"16d1b0623c677fdb7bc555b9cde3e7691cb7601da3c526f43fb022dbdacf3873","first_computed_at":"2026-07-05T11:46:36.419099Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:46:36.419099Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"a4KMgFyB8v4VqNpsLyPy0ZN/tWvZg1Yd77wcGfo0v0+r94kmd5joeuSQRYZrKMFNKMxJ9GQaDRukv1CkdOb7DQ==","signature_status":"signed_v1","signed_at":"2026-07-05T11:46:36.419714Z","signed_message":"canonical_sha256_bytes"},"source_id":"2410.02713","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:d459bee5cd3039247e96a6577e0bff5e9e17ed15087e9874116be134f719a813","sha256:a0cef3bb89205e6588729087b27d96efb9a48460a41a19bd96d34d8feffcac31"],"state_sha256":"731621e74ef896b5e238bda64ca14a2b6b3d1431c4d63adf8ef0553cc95104c6"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"1do5PJZ5r4Rk0VF6VrCHIK5BcHAmYBVKrCDaW/XmM3DaBmmLiYuVwRDArshbVpfobX4SYQybgNNG/mjw8AZVCg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-11T00:06:15.845694Z","bundle_sha256":"bb69db51aa8c6ed7ab95a4d5286ef6534d90758905d4cc5ae75ccffeb847f6dc"}}