{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BPN3KD7B5OP3YY4SZULIQ7D6IX","short_pith_number":"pith:BPN3KD7B","schema_version":"1.0","canonical_sha256":"0bdbb50fe1eb9fbc6392cd16887c7e45e896d832deed49343c0bfde46971b7c6","source":{"kind":"arxiv","id":"2507.22210","version":1},"attestation_state":"computed","paper":{"title":"Scaling and Data Saturation in Protein Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"q-bio.QM","authors_text":"Aviv Spinner, Corey M. Hudson, Erika DeBenedictis","submitted_at":"2025-07-29T20:15:01Z","abstract_excerpt":"Data in biology is redundant, noisy, and sparse. How does the type and scale of available data impact model performance? In this work, we specifically investigate how protein language models (pLMs) scale with increasing pretraining data. We investigate this relationship by measuring the performance of protein function prediction on a suite of pLMs pretrained on yearly snapshots of UniRef100 from 2011 to 2024. We find no evidence of model saturation on this task: performance improves--but not monotonically--with added data, and this trend differs between unsupervised and supervised experiments."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.22210","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"q-bio.QM","submitted_at":"2025-07-29T20:15:01Z","cross_cats_sorted":[],"title_canon_sha256":"cce38aa6caba98c81fa76d588feb81e8bd65566550c468e7eebe9cda236978f7","abstract_canon_sha256":"7513602e791121f063e81d287fa7e25b02c8b45bf17cb617476488a5a171e955"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:45:18.547928Z","signature_b64":"g82V7L0sY9t8WOMebE+sUjshpaT4XoWsZVvfgQHj6LVRZU38gQA83IOPuhv/aKyP0/B4vQ5R1ulCRw+SfaqoDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0bdbb50fe1eb9fbc6392cd16887c7e45e896d832deed49343c0bfde46971b7c6","last_reissued_at":"2026-07-05T11:45:18.547445Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:45:18.547445Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling and Data Saturation in Protein Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"q-bio.QM","authors_text":"Aviv Spinner, Corey M. Hudson, Erika DeBenedictis","submitted_at":"2025-07-29T20:15:01Z","abstract_excerpt":"Data in biology is redundant, noisy, and sparse. How does the type and scale of available data impact model performance? In this work, we specifically investigate how protein language models (pLMs) scale with increasing pretraining data. We investigate this relationship by measuring the performance of protein function prediction on a suite of pLMs pretrained on yearly snapshots of UniRef100 from 2011 to 2024. We find no evidence of model saturation on this task: performance improves--but not monotonically--with added data, and this trend differs between unsupervised and supervised experiments."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.22210","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.22210/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.22210","created_at":"2026-07-05T11:45:18.547504+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.22210v1","created_at":"2026-07-05T11:45:18.547504+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.22210","created_at":"2026-07-05T11:45:18.547504+00:00"},{"alias_kind":"pith_short_12","alias_value":"BPN3KD7B5OP3","created_at":"2026-07-05T11:45:18.547504+00:00"},{"alias_kind":"pith_short_16","alias_value":"BPN3KD7B5OP3YY4S","created_at":"2026-07-05T11:45:18.547504+00:00"},{"alias_kind":"pith_short_8","alias_value":"BPN3KD7B","created_at":"2026-07-05T11:45:18.547504+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BPN3KD7B5OP3YY4SZULIQ7D6IX","json":"https://pith.science/pith/BPN3KD7B5OP3YY4SZULIQ7D6IX.json","graph_json":"https://pith.science/api/pith-number/BPN3KD7B5OP3YY4SZULIQ7D6IX/graph.json","events_json":"https://pith.science/api/pith-number/BPN3KD7B5OP3YY4SZULIQ7D6IX/events.json","paper":"https://pith.science/paper/BPN3KD7B"},"agent_actions":{"view_html":"https://pith.science/pith/BPN3KD7B5OP3YY4SZULIQ7D6IX","download_json":"https://pith.science/pith/BPN3KD7B5OP3YY4SZULIQ7D6IX.json","view_paper":"https://pith.science/paper/BPN3KD7B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.22210&json=true","fetch_graph":"https://pith.science/api/pith-number/BPN3KD7B5OP3YY4SZULIQ7D6IX/graph.json","fetch_events":"https://pith.science/api/pith-number/BPN3KD7B5OP3YY4SZULIQ7D6IX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BPN3KD7B5OP3YY4SZULIQ7D6IX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BPN3KD7B5OP3YY4SZULIQ7D6IX/action/storage_attestation","attest_author":"https://pith.science/pith/BPN3KD7B5OP3YY4SZULIQ7D6IX/action/author_attestation","sign_citation":"https://pith.science/pith/BPN3KD7B5OP3YY4SZULIQ7D6IX/action/citation_signature","submit_replication":"https://pith.science/pith/BPN3KD7B5OP3YY4SZULIQ7D6IX/action/replication_record"}},"created_at":"2026-07-05T11:45:18.547504+00:00","updated_at":"2026-07-05T11:45:18.547504+00:00"}