{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:T2MRSJHMSD4HECPHGCSPYBU3LF","short_pith_number":"pith:T2MRSJHM","schema_version":"1.0","canonical_sha256":"9e991924ec90f87209e730a4fc069b595665d041beea1218198ac589683d9cee","source":{"kind":"arxiv","id":"2412.17759","version":1},"attestation_state":"computed","paper":{"title":"Survey of Large Multimodal Model Datasets, Application Categories and Taxonomy","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.AI","authors_text":"Amit Agarwal, Bhargava Kumar, Hitesh Laxmichand Patel, Ishan Banerjee, Priyaranjan Pattnayak, Srikant Panda, Tejaswini Kumar","submitted_at":"2024-12-23T18:15:19Z","abstract_excerpt":"Multimodal learning, a rapidly evolving field in artificial intelligence, seeks to construct more versatile and robust systems by integrating and analyzing diverse types of data, including text, images, audio, and video. Inspired by the human ability to assimilate information through many senses, this method enables applications such as text-to-video conversion, visual question answering, and image captioning. Recent developments in datasets that support multimodal language models (MLLMs) are highlighted in this overview. Large-scale multimodal datasets are essential because they allow for tho"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.17759","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-12-23T18:15:19Z","cross_cats_sorted":["cs.CV","cs.LG"],"title_canon_sha256":"47d1203884c77515858ff5a7fd3d52cfed00de266ae1e33ba3a8d19447c77a29","abstract_canon_sha256":"8f8fce727e2ba214bee2c039aa2fcd6c619a8b25221d838adcecbd6494a3c8f7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:53:27.018193Z","signature_b64":"kPjF917jjnEPPcW1CLy56KBpmQFuRThUXc5HW0GLHm6WeneB2tdunQJXdBkx1CM5+mAJ5Q+S/Vaw8UKwWw7CAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9e991924ec90f87209e730a4fc069b595665d041beea1218198ac589683d9cee","last_reissued_at":"2026-07-05T09:53:27.017751Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:53:27.017751Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Survey of Large Multimodal Model Datasets, Application Categories and Taxonomy","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.AI","authors_text":"Amit Agarwal, Bhargava Kumar, Hitesh Laxmichand Patel, Ishan Banerjee, Priyaranjan Pattnayak, Srikant Panda, Tejaswini Kumar","submitted_at":"2024-12-23T18:15:19Z","abstract_excerpt":"Multimodal learning, a rapidly evolving field in artificial intelligence, seeks to construct more versatile and robust systems by integrating and analyzing diverse types of data, including text, images, audio, and video. Inspired by the human ability to assimilate information through many senses, this method enables applications such as text-to-video conversion, visual question answering, and image captioning. Recent developments in datasets that support multimodal language models (MLLMs) are highlighted in this overview. Large-scale multimodal datasets are essential because they allow for tho"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.17759","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.17759/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.17759","created_at":"2026-07-05T09:53:27.017806+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.17759v1","created_at":"2026-07-05T09:53:27.017806+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.17759","created_at":"2026-07-05T09:53:27.017806+00:00"},{"alias_kind":"pith_short_12","alias_value":"T2MRSJHMSD4H","created_at":"2026-07-05T09:53:27.017806+00:00"},{"alias_kind":"pith_short_16","alias_value":"T2MRSJHMSD4HECPH","created_at":"2026-07-05T09:53:27.017806+00:00"},{"alias_kind":"pith_short_8","alias_value":"T2MRSJHM","created_at":"2026-07-05T09:53:27.017806+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24702","citing_title":"Do Image-Text Metrics Respect Semantic Invariances?","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02871","citing_title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","ref_index":148,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08366","citing_title":"Scaling-Aware Data Selection for End-to-End Autonomous Driving Systems","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T2MRSJHMSD4HECPHGCSPYBU3LF","json":"https://pith.science/pith/T2MRSJHMSD4HECPHGCSPYBU3LF.json","graph_json":"https://pith.science/api/pith-number/T2MRSJHMSD4HECPHGCSPYBU3LF/graph.json","events_json":"https://pith.science/api/pith-number/T2MRSJHMSD4HECPHGCSPYBU3LF/events.json","paper":"https://pith.science/paper/T2MRSJHM"},"agent_actions":{"view_html":"https://pith.science/pith/T2MRSJHMSD4HECPHGCSPYBU3LF","download_json":"https://pith.science/pith/T2MRSJHMSD4HECPHGCSPYBU3LF.json","view_paper":"https://pith.science/paper/T2MRSJHM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.17759&json=true","fetch_graph":"https://pith.science/api/pith-number/T2MRSJHMSD4HECPHGCSPYBU3LF/graph.json","fetch_events":"https://pith.science/api/pith-number/T2MRSJHMSD4HECPHGCSPYBU3LF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T2MRSJHMSD4HECPHGCSPYBU3LF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T2MRSJHMSD4HECPHGCSPYBU3LF/action/storage_attestation","attest_author":"https://pith.science/pith/T2MRSJHMSD4HECPHGCSPYBU3LF/action/author_attestation","sign_citation":"https://pith.science/pith/T2MRSJHMSD4HECPHGCSPYBU3LF/action/citation_signature","submit_replication":"https://pith.science/pith/T2MRSJHMSD4HECPHGCSPYBU3LF/action/replication_record"}},"created_at":"2026-07-05T09:53:27.017806+00:00","updated_at":"2026-07-05T09:53:27.017806+00:00"}