{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:43PA5OQCTVA2OQZSWI6NQTXNXQ","short_pith_number":"pith:43PA5OQC","schema_version":"1.0","canonical_sha256":"e6de0eba029d41a74332b23cd84eedbc2956383f15f5a1f935392d9f87a1e43a","source":{"kind":"arxiv","id":"2202.07206","version":2},"attestation_state":"computed","paper":{"title":"Impact of Pretraining Term Frequencies on Few-Shot Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Matt Gardner, Robert L. Logan IV, Sameer Singh, Yasaman Razeghi","submitted_at":"2022-02-15T05:43:54Z","abstract_excerpt":"Pretrained Language Models (LMs) have demonstrated ability to perform numerical reasoning by extrapolating from a few examples in few-shot settings. However, the extent to which this extrapolation relies on robust reasoning is unclear. In this paper, we investigate how well these models reason with terms that are less frequent in the pretraining data. In particular, we examine the correlations between the model performance on test instances and the frequency of terms from those instances in the pretraining data. We measure the strength of this correlation for a number of GPT-based language mod"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.07206","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-02-15T05:43:54Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"69ed9f6801b0e39c3378fbef07e278d4b0d0fdafcad2c1cb66a089d5e70f336d","abstract_canon_sha256":"f6d5600aab364739b528d6f7de5e446b2abd417ce584797385877cebefc8b7c1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:51:39.452456Z","signature_b64":"7+d5FYpOXr4cFP4L2EtnkRoVf/UBmwg4t7uKLI0cQu9wBG2BFaHIF/kSGB4lTexwEKea12fuqM5u5/kHSggJBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e6de0eba029d41a74332b23cd84eedbc2956383f15f5a1f935392d9f87a1e43a","last_reissued_at":"2026-07-05T05:51:39.452023Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:51:39.452023Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Impact of Pretraining Term Frequencies on Few-Shot Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Matt Gardner, Robert L. Logan IV, Sameer Singh, Yasaman Razeghi","submitted_at":"2022-02-15T05:43:54Z","abstract_excerpt":"Pretrained Language Models (LMs) have demonstrated ability to perform numerical reasoning by extrapolating from a few examples in few-shot settings. However, the extent to which this extrapolation relies on robust reasoning is unclear. In this paper, we investigate how well these models reason with terms that are less frequent in the pretraining data. In particular, we examine the correlations between the model performance on test instances and the frequency of terms from those instances in the pretraining data. We measure the strength of this correlation for a number of GPT-based language mod"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.07206","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.07206/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.07206","created_at":"2026-07-05T05:51:39.452085+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.07206v2","created_at":"2026-07-05T05:51:39.452085+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.07206","created_at":"2026-07-05T05:51:39.452085+00:00"},{"alias_kind":"pith_short_12","alias_value":"43PA5OQCTVA2","created_at":"2026-07-05T05:51:39.452085+00:00"},{"alias_kind":"pith_short_16","alias_value":"43PA5OQCTVA2OQZS","created_at":"2026-07-05T05:51:39.452085+00:00"},{"alias_kind":"pith_short_8","alias_value":"43PA5OQC","created_at":"2026-07-05T05:51:39.452085+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2204.06745","citing_title":"GPT-NeoX-20B: An Open-Source Autoregressive Language Model","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2502.09741","citing_title":"FoNE: Precise Single-Token Number Embeddings via Fourier Features","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2503.18018","citing_title":"Lost in Cultural Translation: Do LLMs Struggle with Math Across Cultural Contexts?","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03258","citing_title":"The Right Answer, the Wrong Direction: Why Transformers Fail at Counting and How to Fix It","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2405.14782","citing_title":"Lessons from the Trenches on Reproducible Evaluation of Language Models","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2304.01373","citing_title":"Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling","ref_index":195,"is_internal_anchor":false},{"citing_arxiv_id":"2202.12837","citing_title":"Rethinking the Role of Demonstrations: What Makes In-Context Learning Work?","ref_index":233,"is_internal_anchor":false},{"citing_arxiv_id":"2211.09085","citing_title":"Galactica: A Large Language Model for Science","ref_index":227,"is_internal_anchor":false},{"citing_arxiv_id":"2211.09085","citing_title":"Galactica: A Large Language Model for Science","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03258","citing_title":"The Right Answer, the Wrong Direction: Why Transformers Fail at Counting and How to Fix It","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18907","citing_title":"Gradient-Based Program Synthesis with Neurally Interpreted Languages","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2206.07682","citing_title":"Emergent Abilities of Large Language Models","ref_index":72,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/43PA5OQCTVA2OQZSWI6NQTXNXQ","json":"https://pith.science/pith/43PA5OQCTVA2OQZSWI6NQTXNXQ.json","graph_json":"https://pith.science/api/pith-number/43PA5OQCTVA2OQZSWI6NQTXNXQ/graph.json","events_json":"https://pith.science/api/pith-number/43PA5OQCTVA2OQZSWI6NQTXNXQ/events.json","paper":"https://pith.science/paper/43PA5OQC"},"agent_actions":{"view_html":"https://pith.science/pith/43PA5OQCTVA2OQZSWI6NQTXNXQ","download_json":"https://pith.science/pith/43PA5OQCTVA2OQZSWI6NQTXNXQ.json","view_paper":"https://pith.science/paper/43PA5OQC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.07206&json=true","fetch_graph":"https://pith.science/api/pith-number/43PA5OQCTVA2OQZSWI6NQTXNXQ/graph.json","fetch_events":"https://pith.science/api/pith-number/43PA5OQCTVA2OQZSWI6NQTXNXQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/43PA5OQCTVA2OQZSWI6NQTXNXQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/43PA5OQCTVA2OQZSWI6NQTXNXQ/action/storage_attestation","attest_author":"https://pith.science/pith/43PA5OQCTVA2OQZSWI6NQTXNXQ/action/author_attestation","sign_citation":"https://pith.science/pith/43PA5OQCTVA2OQZSWI6NQTXNXQ/action/citation_signature","submit_replication":"https://pith.science/pith/43PA5OQCTVA2OQZSWI6NQTXNXQ/action/replication_record"}},"created_at":"2026-07-05T05:51:39.452085+00:00","updated_at":"2026-07-05T05:51:39.452085+00:00"}