{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HJMK2MNQLJG7ILGE5HYZXYON35","short_pith_number":"pith:HJMK2MNQ","schema_version":"1.0","canonical_sha256":"3a58ad31b05a4df42cc4e9f19be1cddf4545feab32e6e7d8b7cc60ff3c282113","source":{"kind":"arxiv","id":"2305.14947","version":2},"attestation_state":"computed","paper":{"title":"How Predictable Are Large Language Model Capabilities? A Case Study on BIG-bench","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Harvey Yiyun Fu, Qinyuan Ye, Robin Jia, Xiang Ren","submitted_at":"2023-05-24T09:35:34Z","abstract_excerpt":"We investigate the predictability of large language model (LLM) capabilities: given records of past experiments using different model families, numbers of parameters, tasks, and numbers of in-context examples, can we accurately predict LLM performance on new experiment configurations? Answering this question has practical implications for LLM users (e.g., deciding which models to try), developers (e.g., prioritizing evaluation on representative tasks), and the research community (e.g., identifying hard-to-predict capabilities that warrant further investigation).\n  We study the performance pred"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.14947","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-24T09:35:34Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"5a817efaae28da0d31ef14b968c678d1e73c3a07fb0b75625d160f0341b149a2","abstract_canon_sha256":"52cbcc6cee3d276a75d5d0df43eb89b9417ecea33c1e0f0581dd03306d6b3dbf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:07:10.547100Z","signature_b64":"R7PmDEe/f0SWfFshsgcPJzP+45KtviwdgQHwTbis3Us0xKZJqkPSDEhMFVukGwTi8drg5LiDU4+WaeySBTh6CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3a58ad31b05a4df42cc4e9f19be1cddf4545feab32e6e7d8b7cc60ff3c282113","last_reissued_at":"2026-07-05T07:07:10.546621Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:07:10.546621Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Predictable Are Large Language Model Capabilities? A Case Study on BIG-bench","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Harvey Yiyun Fu, Qinyuan Ye, Robin Jia, Xiang Ren","submitted_at":"2023-05-24T09:35:34Z","abstract_excerpt":"We investigate the predictability of large language model (LLM) capabilities: given records of past experiments using different model families, numbers of parameters, tasks, and numbers of in-context examples, can we accurately predict LLM performance on new experiment configurations? Answering this question has practical implications for LLM users (e.g., deciding which models to try), developers (e.g., prioritizing evaluation on representative tasks), and the research community (e.g., identifying hard-to-predict capabilities that warrant further investigation).\n  We study the performance pred"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.14947","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.14947/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.14947","created_at":"2026-07-05T07:07:10.546669+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.14947v2","created_at":"2026-07-05T07:07:10.546669+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.14947","created_at":"2026-07-05T07:07:10.546669+00:00"},{"alias_kind":"pith_short_12","alias_value":"HJMK2MNQLJG7","created_at":"2026-07-05T07:07:10.546669+00:00"},{"alias_kind":"pith_short_16","alias_value":"HJMK2MNQLJG7ILGE","created_at":"2026-07-05T07:07:10.546669+00:00"},{"alias_kind":"pith_short_8","alias_value":"HJMK2MNQ","created_at":"2026-07-05T07:07:10.546669+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.13216","citing_title":"Capability Salience Vector: Fine-grained Alignment of Loss and Capabilities for Downstream Task Scaling Law","ref_index":35,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HJMK2MNQLJG7ILGE5HYZXYON35","json":"https://pith.science/pith/HJMK2MNQLJG7ILGE5HYZXYON35.json","graph_json":"https://pith.science/api/pith-number/HJMK2MNQLJG7ILGE5HYZXYON35/graph.json","events_json":"https://pith.science/api/pith-number/HJMK2MNQLJG7ILGE5HYZXYON35/events.json","paper":"https://pith.science/paper/HJMK2MNQ"},"agent_actions":{"view_html":"https://pith.science/pith/HJMK2MNQLJG7ILGE5HYZXYON35","download_json":"https://pith.science/pith/HJMK2MNQLJG7ILGE5HYZXYON35.json","view_paper":"https://pith.science/paper/HJMK2MNQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.14947&json=true","fetch_graph":"https://pith.science/api/pith-number/HJMK2MNQLJG7ILGE5HYZXYON35/graph.json","fetch_events":"https://pith.science/api/pith-number/HJMK2MNQLJG7ILGE5HYZXYON35/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HJMK2MNQLJG7ILGE5HYZXYON35/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HJMK2MNQLJG7ILGE5HYZXYON35/action/storage_attestation","attest_author":"https://pith.science/pith/HJMK2MNQLJG7ILGE5HYZXYON35/action/author_attestation","sign_citation":"https://pith.science/pith/HJMK2MNQLJG7ILGE5HYZXYON35/action/citation_signature","submit_replication":"https://pith.science/pith/HJMK2MNQLJG7ILGE5HYZXYON35/action/replication_record"}},"created_at":"2026-07-05T07:07:10.546669+00:00","updated_at":"2026-07-05T07:07:10.546669+00:00"}