{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Y7HRIR2NQTYSAVNEGQGCBNULVE","short_pith_number":"pith:Y7HRIR2N","schema_version":"1.0","canonical_sha256":"c7cf14474d84f12055a4340c20b68ba932bc3bdbf541f1ac89d34b884379ebb7","source":{"kind":"arxiv","id":"2407.13773","version":1},"attestation_state":"computed","paper":{"title":"OpenDataLab: Empowering General Artificial Intelligence with Open Datasets","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.DL","authors_text":"Bin Wang, Chao Xu, Conghui He, Dahua Lin, Wei Li, Zhenjiang Jin","submitted_at":"2024-06-04T10:42:01Z","abstract_excerpt":"The advancement of artificial intelligence (AI) hinges on the quality and accessibility of data, yet the current fragmentation and variability of data sources hinder efficient data utilization. The dispersion of data sources and diversity of data formats often lead to inefficiencies in data retrieval and processing, significantly impeding the progress of AI research and applications. To address these challenges, this paper introduces OpenDataLab, a platform designed to bridge the gap between diverse data sources and the need for unified data processing. OpenDataLab integrates a wide range of o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.13773","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DL","submitted_at":"2024-06-04T10:42:01Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e41d62c0814272d90a28ba5cd54ca0b2f1562b1b1e45ca20ba9d1a14597a01cd","abstract_canon_sha256":"5d3b037ed3bf4e80b67e1bf0b5b136f07ab883a2891bd6374e07702cf4012edd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:45:53.075581Z","signature_b64":"aMxgedSJKSFiO6uszfCO8QY6YPnjHAZb5w5o3hkO5a7kRStF4sb9e/dFDtsdaEnR+FxmFtfRIrmF3C3sRtKCCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c7cf14474d84f12055a4340c20b68ba932bc3bdbf541f1ac89d34b884379ebb7","last_reissued_at":"2026-07-05T08:45:53.075077Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:45:53.075077Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OpenDataLab: Empowering General Artificial Intelligence with Open Datasets","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.DL","authors_text":"Bin Wang, Chao Xu, Conghui He, Dahua Lin, Wei Li, Zhenjiang Jin","submitted_at":"2024-06-04T10:42:01Z","abstract_excerpt":"The advancement of artificial intelligence (AI) hinges on the quality and accessibility of data, yet the current fragmentation and variability of data sources hinder efficient data utilization. The dispersion of data sources and diversity of data formats often lead to inefficiencies in data retrieval and processing, significantly impeding the progress of AI research and applications. To address these challenges, this paper introduces OpenDataLab, a platform designed to bridge the gap between diverse data sources and the need for unified data processing. OpenDataLab integrates a wide range of o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.13773","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.13773/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.13773","created_at":"2026-07-05T08:45:53.075137+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.13773v1","created_at":"2026-07-05T08:45:53.075137+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.13773","created_at":"2026-07-05T08:45:53.075137+00:00"},{"alias_kind":"pith_short_12","alias_value":"Y7HRIR2NQTYS","created_at":"2026-07-05T08:45:53.075137+00:00"},{"alias_kind":"pith_short_16","alias_value":"Y7HRIR2NQTYSAVNE","created_at":"2026-07-05T08:45:53.075137+00:00"},{"alias_kind":"pith_short_8","alias_value":"Y7HRIR2N","created_at":"2026-07-05T08:45:53.075137+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08376","citing_title":"RiskNet: A large-scale dataset of AI risk incidents from news with alignment and multi-dimensional annotations","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.14096","citing_title":"A New Multi-Domain Benchmark for Micro-Action Recognition and Detection","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2504.11101","citing_title":"Consensus Entropy: Harnessing Multi-VLM Agreement for Self-Verifying and Self-Improving OCR","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18799","citing_title":"ReCrit: Transition-Aware Reinforcement Learning for Scientific Critic Reasoning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09927","citing_title":"Information Extraction of Nested Complex Structure of Quantum Cascade Lasers via Large Language Models","ref_index":48,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Y7HRIR2NQTYSAVNEGQGCBNULVE","json":"https://pith.science/pith/Y7HRIR2NQTYSAVNEGQGCBNULVE.json","graph_json":"https://pith.science/api/pith-number/Y7HRIR2NQTYSAVNEGQGCBNULVE/graph.json","events_json":"https://pith.science/api/pith-number/Y7HRIR2NQTYSAVNEGQGCBNULVE/events.json","paper":"https://pith.science/paper/Y7HRIR2N"},"agent_actions":{"view_html":"https://pith.science/pith/Y7HRIR2NQTYSAVNEGQGCBNULVE","download_json":"https://pith.science/pith/Y7HRIR2NQTYSAVNEGQGCBNULVE.json","view_paper":"https://pith.science/paper/Y7HRIR2N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.13773&json=true","fetch_graph":"https://pith.science/api/pith-number/Y7HRIR2NQTYSAVNEGQGCBNULVE/graph.json","fetch_events":"https://pith.science/api/pith-number/Y7HRIR2NQTYSAVNEGQGCBNULVE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Y7HRIR2NQTYSAVNEGQGCBNULVE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Y7HRIR2NQTYSAVNEGQGCBNULVE/action/storage_attestation","attest_author":"https://pith.science/pith/Y7HRIR2NQTYSAVNEGQGCBNULVE/action/author_attestation","sign_citation":"https://pith.science/pith/Y7HRIR2NQTYSAVNEGQGCBNULVE/action/citation_signature","submit_replication":"https://pith.science/pith/Y7HRIR2NQTYSAVNEGQGCBNULVE/action/replication_record"}},"created_at":"2026-07-05T08:45:53.075137+00:00","updated_at":"2026-07-05T08:45:53.075137+00:00"}