{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:MIUIVO5AWTIDOZNMK3VSAUCGYJ","short_pith_number":"pith:MIUIVO5A","schema_version":"1.0","canonical_sha256":"62288abba0b4d03765ac56eb205046c25553ea8d8d5b9567b9a3589061e66dd0","source":{"kind":"arxiv","id":"2102.09690","version":2},"attestation_state":"computed","paper":{"title":"Calibrate Before Use: Improving Few-Shot Performance of Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Dan Klein, Eric Wallace, Sameer Singh, Shi Feng, Tony Z. Zhao","submitted_at":"2021-02-19T00:23:59Z","abstract_excerpt":"GPT-3 can perform numerous tasks when provided a natural language prompt that contains a few training examples. We show that this type of few-shot learning can be unstable: the choice of prompt format, training examples, and even the order of the training examples can cause accuracy to vary from near chance to near state-of-the-art. We demonstrate that this instability arises from the bias of language models towards predicting certain answers, e.g., those that are placed near the end of the prompt or are common in the pre-training data. To mitigate this, we first estimate the model's bias towa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2102.09690","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-02-19T00:23:59Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"32c37c263561c589f8d9e8d801b07f170484345774156101bb83ffc30f974ace","abstract_canon_sha256":"cee955a412abad581b0fe9ed38f22052422219d4c60092ba564b1aa63cc6dacb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:48:21.550044Z","signature_b64":"B3Xa2xQZbgHqV21MPWp4HwnUZEy1tDycleRZ4vcY38YpTRU7k5UZQxjHQhArhDEhSEDAuaAGmcXPV85d1yiBAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"62288abba0b4d03765ac56eb205046c25553ea8d8d5b9567b9a3589061e66dd0","last_reissued_at":"2026-07-05T02:48:21.549626Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:48:21.549626Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Calibrate Before Use: Improving Few-Shot Performance of Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Dan Klein, Eric Wallace, Sameer Singh, Shi Feng, Tony Z. Zhao","submitted_at":"2021-02-19T00:23:59Z","abstract_excerpt":"GPT-3 can perform numerous tasks when provided a natural language prompt that contains a few training examples. We show that this type of few-shot learning can be unstable: the choice of prompt format, training examples, and even the order of the training examples can cause accuracy to vary from near chance to near state-of-the-art. We demonstrate that this instability arises from the bias of language models towards predicting certain answers, e.g., those that are placed near the end of the prompt or are common in the pre-training data. To mitigate this, we first estimate the model's bias towa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.09690","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.09690/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2102.09690","created_at":"2026-07-05T02:48:21.549683+00:00"},{"alias_kind":"arxiv_version","alias_value":"2102.09690v2","created_at":"2026-07-05T02:48:21.549683+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.09690","created_at":"2026-07-05T02:48:21.549683+00:00"},{"alias_kind":"pith_short_12","alias_value":"MIUIVO5AWTID","created_at":"2026-07-05T02:48:21.549683+00:00"},{"alias_kind":"pith_short_16","alias_value":"MIUIVO5AWTIDOZNM","created_at":"2026-07-05T02:48:21.549683+00:00"},{"alias_kind":"pith_short_8","alias_value":"MIUIVO5A","created_at":"2026-07-05T02:48:21.549683+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22470","citing_title":"PRIME: Evaluating Prompt Resolution Under Incompatible Instructions in LLMs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07479","citing_title":"Supervision versus Demonstration-Based In-Context Learning for Multiword Expression Classification","ref_index":219,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22714","citing_title":"AMEL: Accumulated Message Effects on LLM Judgments","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24718","citing_title":"The Tokenizer Tax Across 25 European Languages: Domain Invariance, Cross-Lingual Few-Shot Effects, and the Ukrainian Penalty","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29170","citing_title":"UA-Legal-Bench: A Benchmark for Evaluating Large Language Models on Ukrainian Legal Reasoning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22714","citing_title":"AMEL: Accumulated Message Effects on LLM Judgments","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2603.11689","citing_title":"Explicit Logic Channel for Validation and Enhancement of MLLMs on Zero-Shot Tasks","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2308.03958","citing_title":"Simple synthetic data reduces sycophancy in large language models","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2212.03827","citing_title":"Discovering Latent Knowledge in Language Models Without Supervision","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24678","citing_title":"Leveraging LLMs for Multi-File DSL Code Generation: An Industrial Case Study","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23371","citing_title":"When Context Sticks: Studying Interference in In-Context Learning","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2206.07682","citing_title":"Emergent Abilities of Large Language Models","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2201.11903","citing_title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15547","citing_title":"Consistency Analysis of Sentiment Predictions using Syntactic & Semantic Context Assessment Summarization (SSAS)","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MIUIVO5AWTIDOZNMK3VSAUCGYJ","json":"https://pith.science/pith/MIUIVO5AWTIDOZNMK3VSAUCGYJ.json","graph_json":"https://pith.science/api/pith-number/MIUIVO5AWTIDOZNMK3VSAUCGYJ/graph.json","events_json":"https://pith.science/api/pith-number/MIUIVO5AWTIDOZNMK3VSAUCGYJ/events.json","paper":"https://pith.science/paper/MIUIVO5A"},"agent_actions":{"view_html":"https://pith.science/pith/MIUIVO5AWTIDOZNMK3VSAUCGYJ","download_json":"https://pith.science/pith/MIUIVO5AWTIDOZNMK3VSAUCGYJ.json","view_paper":"https://pith.science/paper/MIUIVO5A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2102.09690&json=true","fetch_graph":"https://pith.science/api/pith-number/MIUIVO5AWTIDOZNMK3VSAUCGYJ/graph.json","fetch_events":"https://pith.science/api/pith-number/MIUIVO5AWTIDOZNMK3VSAUCGYJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MIUIVO5AWTIDOZNMK3VSAUCGYJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MIUIVO5AWTIDOZNMK3VSAUCGYJ/action/storage_attestation","attest_author":"https://pith.science/pith/MIUIVO5AWTIDOZNMK3VSAUCGYJ/action/author_attestation","sign_citation":"https://pith.science/pith/MIUIVO5AWTIDOZNMK3VSAUCGYJ/action/citation_signature","submit_replication":"https://pith.science/pith/MIUIVO5AWTIDOZNMK3VSAUCGYJ/action/replication_record"}},"created_at":"2026-07-05T02:48:21.549683+00:00","updated_at":"2026-07-05T02:48:21.549683+00:00"}