{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4HUZHGWV6AA6EU6CA7OYBTG3JP","short_pith_number":"pith:4HUZHGWV","schema_version":"1.0","canonical_sha256":"e1e9939ad5f001e253c207dd80ccdb4bd7d1408d76db9de7c9573b464133d1ab","source":{"kind":"arxiv","id":"2404.06757","version":1},"attestation_state":"computed","paper":{"title":"Language Generation in the Limit","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.DS","authors_text":"Jon Kleinberg, Sendhil Mullainathan","submitted_at":"2024-04-10T05:53:25Z","abstract_excerpt":"Although current large language models are complex, the most basic specifications of the underlying language generation problem itself are simple to state: given a finite set of training samples from an unknown language, produce valid new strings from the language that don't already appear in the training data. Here we ask what we can conclude about language generation using only this specification, without further assumptions. In particular, suppose that an adversary enumerates the strings of an unknown target language L that is known only to come from one of a possibly infinite list of candi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.06757","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DS","submitted_at":"2024-04-10T05:53:25Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"c52f59036f93b73c236187be177c35a79537d36a73e31682146c629d09359381","abstract_canon_sha256":"f66015dd1675051d955f41310d37ce9967a70882ba209eb3c68627e1364e193b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:06:29.681905Z","signature_b64":"arXd+xTSA8h5hBpajMHfF+t1mBBDaRj4UDCBr/cOjQAjwpP00gQnWMxCMdVt+UACUpWMBT9olDMIPtpVYqBTBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e1e9939ad5f001e253c207dd80ccdb4bd7d1408d76db9de7c9573b464133d1ab","last_reissued_at":"2026-07-05T08:06:29.681443Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:06:29.681443Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Language Generation in the Limit","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.DS","authors_text":"Jon Kleinberg, Sendhil Mullainathan","submitted_at":"2024-04-10T05:53:25Z","abstract_excerpt":"Although current large language models are complex, the most basic specifications of the underlying language generation problem itself are simple to state: given a finite set of training samples from an unknown language, produce valid new strings from the language that don't already appear in the training data. Here we ask what we can conclude about language generation using only this specification, without further assumptions. In particular, suppose that an adversary enumerates the strings of an unknown target language L that is known only to come from one of a possibly infinite list of candi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.06757","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.06757/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.06757","created_at":"2026-07-05T08:06:29.681501+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.06757v1","created_at":"2026-07-05T08:06:29.681501+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.06757","created_at":"2026-07-05T08:06:29.681501+00:00"},{"alias_kind":"pith_short_12","alias_value":"4HUZHGWV6AA6","created_at":"2026-07-05T08:06:29.681501+00:00"},{"alias_kind":"pith_short_16","alias_value":"4HUZHGWV6AA6EU6C","created_at":"2026-07-05T08:06:29.681501+00:00"},{"alias_kind":"pith_short_8","alias_value":"4HUZHGWV","created_at":"2026-07-05T08:06:29.681501+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28354","citing_title":"Generating in the Limit with Infinitely Many Hallucinations","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30324","citing_title":"On Language Generation in the Limit with Bounded Memory","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13934","citing_title":"Why Code, Why Now: An Information-Theoretic Perspective on the Limits of Machine Learning","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4HUZHGWV6AA6EU6CA7OYBTG3JP","json":"https://pith.science/pith/4HUZHGWV6AA6EU6CA7OYBTG3JP.json","graph_json":"https://pith.science/api/pith-number/4HUZHGWV6AA6EU6CA7OYBTG3JP/graph.json","events_json":"https://pith.science/api/pith-number/4HUZHGWV6AA6EU6CA7OYBTG3JP/events.json","paper":"https://pith.science/paper/4HUZHGWV"},"agent_actions":{"view_html":"https://pith.science/pith/4HUZHGWV6AA6EU6CA7OYBTG3JP","download_json":"https://pith.science/pith/4HUZHGWV6AA6EU6CA7OYBTG3JP.json","view_paper":"https://pith.science/paper/4HUZHGWV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.06757&json=true","fetch_graph":"https://pith.science/api/pith-number/4HUZHGWV6AA6EU6CA7OYBTG3JP/graph.json","fetch_events":"https://pith.science/api/pith-number/4HUZHGWV6AA6EU6CA7OYBTG3JP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4HUZHGWV6AA6EU6CA7OYBTG3JP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4HUZHGWV6AA6EU6CA7OYBTG3JP/action/storage_attestation","attest_author":"https://pith.science/pith/4HUZHGWV6AA6EU6CA7OYBTG3JP/action/author_attestation","sign_citation":"https://pith.science/pith/4HUZHGWV6AA6EU6CA7OYBTG3JP/action/citation_signature","submit_replication":"https://pith.science/pith/4HUZHGWV6AA6EU6CA7OYBTG3JP/action/replication_record"}},"created_at":"2026-07-05T08:06:29.681501+00:00","updated_at":"2026-07-05T08:06:29.681501+00:00"}