{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:K2VXUGNLT56WEDXBTV2GZ7AMQD","short_pith_number":"pith:K2VXUGNL","schema_version":"1.0","canonical_sha256":"56ab7a19ab9f7d620ee19d746cfc0c80e1bca2caf615e798afc10bf6a6f6d682","source":{"kind":"arxiv","id":"2501.13779","version":2},"attestation_state":"computed","paper":{"title":"Not Every AI Problem is a Data Problem: We Should Be Intentional About Data Scaling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Natasha Noy, Nino Scherrer, Tanya Rodchenko","submitted_at":"2025-01-23T15:58:14Z","abstract_excerpt":"While Large Language Models require more and more data to train and scale, rather than looking for any data to acquire, we should consider what types of tasks are more likely to benefit from data scaling. We should be intentional in our data acquisition. We argue that the shape of the data itself, such as its compositional and structural patterns, informs which tasks to prioritize in data scaling, and shapes the development of the next generation of compute paradigms for tasks where data scaling is inefficient, or even insufficient."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.13779","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-23T15:58:14Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"86f94a12140fb795297baefac68ce20b0a823332fe4e45046caf84a57d76db7c","abstract_canon_sha256":"110f6bc41a140e82b68c1b09b7d56154397e1503d03043268742f24d19105101"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:52.824640Z","signature_b64":"7R8oAVUVVRES7n9OVOwgJn3ZwyYly5tPjcHkSQbCWSCPwUZ6ZF68zCzh9B0j4VM94RARKJV0qCsmsifn65YwCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"56ab7a19ab9f7d620ee19d746cfc0c80e1bca2caf615e798afc10bf6a6f6d682","last_reissued_at":"2026-07-05T11:14:52.824232Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:52.824232Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Not Every AI Problem is a Data Problem: We Should Be Intentional About Data Scaling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Natasha Noy, Nino Scherrer, Tanya Rodchenko","submitted_at":"2025-01-23T15:58:14Z","abstract_excerpt":"While Large Language Models require more and more data to train and scale, rather than looking for any data to acquire, we should consider what types of tasks are more likely to benefit from data scaling. We should be intentional in our data acquisition. We argue that the shape of the data itself, such as its compositional and structural patterns, informs which tasks to prioritize in data scaling, and shapes the development of the next generation of compute paradigms for tasks where data scaling is inefficient, or even insufficient."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.13779","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.13779/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.13779","created_at":"2026-07-05T11:14:52.824289+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.13779v2","created_at":"2026-07-05T11:14:52.824289+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.13779","created_at":"2026-07-05T11:14:52.824289+00:00"},{"alias_kind":"pith_short_12","alias_value":"K2VXUGNLT56W","created_at":"2026-07-05T11:14:52.824289+00:00"},{"alias_kind":"pith_short_16","alias_value":"K2VXUGNLT56WEDXB","created_at":"2026-07-05T11:14:52.824289+00:00"},{"alias_kind":"pith_short_8","alias_value":"K2VXUGNL","created_at":"2026-07-05T11:14:52.824289+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.05233","citing_title":"MesaNet: Sequence Modeling by Locally Optimal Test-Time Training","ref_index":91,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K2VXUGNLT56WEDXBTV2GZ7AMQD","json":"https://pith.science/pith/K2VXUGNLT56WEDXBTV2GZ7AMQD.json","graph_json":"https://pith.science/api/pith-number/K2VXUGNLT56WEDXBTV2GZ7AMQD/graph.json","events_json":"https://pith.science/api/pith-number/K2VXUGNLT56WEDXBTV2GZ7AMQD/events.json","paper":"https://pith.science/paper/K2VXUGNL"},"agent_actions":{"view_html":"https://pith.science/pith/K2VXUGNLT56WEDXBTV2GZ7AMQD","download_json":"https://pith.science/pith/K2VXUGNLT56WEDXBTV2GZ7AMQD.json","view_paper":"https://pith.science/paper/K2VXUGNL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.13779&json=true","fetch_graph":"https://pith.science/api/pith-number/K2VXUGNLT56WEDXBTV2GZ7AMQD/graph.json","fetch_events":"https://pith.science/api/pith-number/K2VXUGNLT56WEDXBTV2GZ7AMQD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K2VXUGNLT56WEDXBTV2GZ7AMQD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K2VXUGNLT56WEDXBTV2GZ7AMQD/action/storage_attestation","attest_author":"https://pith.science/pith/K2VXUGNLT56WEDXBTV2GZ7AMQD/action/author_attestation","sign_citation":"https://pith.science/pith/K2VXUGNLT56WEDXBTV2GZ7AMQD/action/citation_signature","submit_replication":"https://pith.science/pith/K2VXUGNLT56WEDXBTV2GZ7AMQD/action/replication_record"}},"created_at":"2026-07-05T11:14:52.824289+00:00","updated_at":"2026-07-05T11:14:52.824289+00:00"}