{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AMAL6N67DUVDWH7UT5QFKCVIHP","short_pith_number":"pith:AMAL6N67","schema_version":"1.0","canonical_sha256":"0300bf37df1d2a3b1ff49f60550aa83be61c147b390cf0ecdb289105befbd6c0","source":{"kind":"arxiv","id":"2411.19951","version":5},"attestation_state":"computed","paper":{"title":"Sparrow: Data-Efficient Video-LLM with Text-to-Image Augmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Caifeng Shan, Chaoyou Fu, Chunjiang Ge, Enhong Chen, Shukang Yin, Sirui Zhao, Tong Xu, Yan Yang, Yongdong Luo, Yuhan Dai","submitted_at":"2024-11-29T18:59:54Z","abstract_excerpt":"Recent years have seen the success of Multimodal Large Language Models (MLLMs) in the domain of vision understanding. The success of these models can largely be attributed to the dominant scaling law, which states that larger parameter sizes and data volumes contribute to better performance. Notably, data scaling has been primarily driven by automatic data pipelines, which focus on the self-instruction of LLMs. The paradigm has been taken for granted for quite some time, but the study of the effectiveness of scaling with these data has been neglected for a long time. In this context, this work"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.19951","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-29T18:59:54Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"c628d16e627d6e5c0e8af701cb0cb2e1d8159b32e5a0796435fd2ab739ff24ca","abstract_canon_sha256":"011fbc285ed6df6dba2f113cac0a216bd869adfd674fcd45dcd162e7cddff50b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:40:51.091962Z","signature_b64":"wxGccOOxlLussEih1fIzwpCo3rUuO2EhYGMKKb+6wPpjA5sAxxdhemM+Wo8VIKJklIHKLp0Yk6KT1em4b3kWAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0300bf37df1d2a3b1ff49f60550aa83be61c147b390cf0ecdb289105befbd6c0","last_reissued_at":"2026-07-05T11:40:51.091412Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:40:51.091412Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sparrow: Data-Efficient Video-LLM with Text-to-Image Augmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Caifeng Shan, Chaoyou Fu, Chunjiang Ge, Enhong Chen, Shukang Yin, Sirui Zhao, Tong Xu, Yan Yang, Yongdong Luo, Yuhan Dai","submitted_at":"2024-11-29T18:59:54Z","abstract_excerpt":"Recent years have seen the success of Multimodal Large Language Models (MLLMs) in the domain of vision understanding. The success of these models can largely be attributed to the dominant scaling law, which states that larger parameter sizes and data volumes contribute to better performance. Notably, data scaling has been primarily driven by automatic data pipelines, which focus on the self-instruction of LLMs. The paradigm has been taken for granted for quite some time, but the study of the effectiveness of scaling with these data has been neglected for a long time. In this context, this work"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.19951","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.19951/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.19951","created_at":"2026-07-05T11:40:51.091472+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.19951v5","created_at":"2026-07-05T11:40:51.091472+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.19951","created_at":"2026-07-05T11:40:51.091472+00:00"},{"alias_kind":"pith_short_12","alias_value":"AMAL6N67DUVD","created_at":"2026-07-05T11:40:51.091472+00:00"},{"alias_kind":"pith_short_16","alias_value":"AMAL6N67DUVDWH7U","created_at":"2026-07-05T11:40:51.091472+00:00"},{"alias_kind":"pith_short_8","alias_value":"AMAL6N67","created_at":"2026-07-05T11:40:51.091472+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.06355","citing_title":"LLMs as World Models: Data-Driven and Human-Centered Pre-Event Simulation for Disaster Impact Assessment","ref_index":52,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AMAL6N67DUVDWH7UT5QFKCVIHP","json":"https://pith.science/pith/AMAL6N67DUVDWH7UT5QFKCVIHP.json","graph_json":"https://pith.science/api/pith-number/AMAL6N67DUVDWH7UT5QFKCVIHP/graph.json","events_json":"https://pith.science/api/pith-number/AMAL6N67DUVDWH7UT5QFKCVIHP/events.json","paper":"https://pith.science/paper/AMAL6N67"},"agent_actions":{"view_html":"https://pith.science/pith/AMAL6N67DUVDWH7UT5QFKCVIHP","download_json":"https://pith.science/pith/AMAL6N67DUVDWH7UT5QFKCVIHP.json","view_paper":"https://pith.science/paper/AMAL6N67","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.19951&json=true","fetch_graph":"https://pith.science/api/pith-number/AMAL6N67DUVDWH7UT5QFKCVIHP/graph.json","fetch_events":"https://pith.science/api/pith-number/AMAL6N67DUVDWH7UT5QFKCVIHP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AMAL6N67DUVDWH7UT5QFKCVIHP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AMAL6N67DUVDWH7UT5QFKCVIHP/action/storage_attestation","attest_author":"https://pith.science/pith/AMAL6N67DUVDWH7UT5QFKCVIHP/action/author_attestation","sign_citation":"https://pith.science/pith/AMAL6N67DUVDWH7UT5QFKCVIHP/action/citation_signature","submit_replication":"https://pith.science/pith/AMAL6N67DUVDWH7UT5QFKCVIHP/action/replication_record"}},"created_at":"2026-07-05T11:40:51.091472+00:00","updated_at":"2026-07-05T11:40:51.091472+00:00"}