{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3WYKEHWNJAKVYZ7EYVRMWYT7XM","short_pith_number":"pith:3WYKEHWN","schema_version":"1.0","canonical_sha256":"ddb0a21ecd48155c67e4c562cb627fbb03cec454b6ad3702ae99c7e73c40fd4e","source":{"kind":"arxiv","id":"2406.16838","version":2},"attestation_state":"computed","paper":{"title":"From Decoding to Meta-Generation: Inference-time Algorithms for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Alex Xie, Amanda Bertsch, Graham Neubig, Hailey Schoelkopf, Ilia Kulikov, Matthew Finlayson, Sean Welleck, Zaid Harchaoui","submitted_at":"2024-06-24T17:45:59Z","abstract_excerpt":"One of the most striking findings in modern research on large language models (LLMs) is that scaling up compute during training leads to better results. However, less attention has been given to the benefits of scaling compute during inference. This survey focuses on these inference-time approaches. We explore three areas under a unified mathematical formalism: token-level generation algorithms, meta-generation algorithms, and efficient generation. Token-level generation algorithms, often called decoding algorithms, operate by sampling a single token at a time or constructing a token-level sea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.16838","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-24T17:45:59Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"9d3e1ae5d6b56becc18af63994321e7934f299af1dca9d3dcd4a725f15291b68","abstract_canon_sha256":"a487220e1494c4bdc6c0d5b32a11d82f1a9ea18669a7009d64f2cadf1e9b0678"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:38:15.170306Z","signature_b64":"95oy2hxLdRuA6PVoUmVaRJmKqk47WGR6W4c5NbtWFsVfaTy3mAJ0sinSiojJIRVn9liypyWoJ/qgk2gziO+7Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ddb0a21ecd48155c67e4c562cb627fbb03cec454b6ad3702ae99c7e73c40fd4e","last_reissued_at":"2026-07-05T09:38:15.169800Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:38:15.169800Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"From Decoding to Meta-Generation: Inference-time Algorithms for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Alex Xie, Amanda Bertsch, Graham Neubig, Hailey Schoelkopf, Ilia Kulikov, Matthew Finlayson, Sean Welleck, Zaid Harchaoui","submitted_at":"2024-06-24T17:45:59Z","abstract_excerpt":"One of the most striking findings in modern research on large language models (LLMs) is that scaling up compute during training leads to better results. However, less attention has been given to the benefits of scaling compute during inference. This survey focuses on these inference-time approaches. We explore three areas under a unified mathematical formalism: token-level generation algorithms, meta-generation algorithms, and efficient generation. Token-level generation algorithms, often called decoding algorithms, operate by sampling a single token at a time or constructing a token-level sea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.16838","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.16838/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.16838","created_at":"2026-07-05T09:38:15.169863+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.16838v2","created_at":"2026-07-05T09:38:15.169863+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.16838","created_at":"2026-07-05T09:38:15.169863+00:00"},{"alias_kind":"pith_short_12","alias_value":"3WYKEHWNJAKV","created_at":"2026-07-05T09:38:15.169863+00:00"},{"alias_kind":"pith_short_16","alias_value":"3WYKEHWNJAKVYZ7E","created_at":"2026-07-05T09:38:15.169863+00:00"},{"alias_kind":"pith_short_8","alias_value":"3WYKEHWN","created_at":"2026-07-05T09:38:15.169863+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01571","citing_title":"Geometric Signatures of Reasoning: A Spectral Perspective on Task Hardness","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12384","citing_title":"APPO: Agentic Procedural Policy Optimization","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2509.01728","citing_title":"Constrained Decoding for Safe Robot Navigation Foundation Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2505.15134","citing_title":"The Unreasonable Effectiveness of Entropy Minimization in LLM Reasoning","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04944","citing_title":"Inclusion-of-Thoughts: Mitigating Preference Instability via Purifying the Decision Space","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2409.02813","citing_title":"MMMU-Pro: A More Robust Multi-discipline Multimodal Understanding Benchmark","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11361","citing_title":"The tractability landscape of diffusion alignment: regularization, rewards, and computational primitives","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03356","citing_title":"POSTCONDBENCH: Benchmarking Correctness and Completeness in Formal Postcondition Inference","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04559","citing_title":"Beyond Static Best-of-N: Bayesian List-wise Alignment for LLM-based Recommendation","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10827","citing_title":"Your Model Diversity, Not Method, Determines Reasoning Strategy","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10667","citing_title":"Learning and Enforcing Context-Sensitive Control for LLMs","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3WYKEHWNJAKVYZ7EYVRMWYT7XM","json":"https://pith.science/pith/3WYKEHWNJAKVYZ7EYVRMWYT7XM.json","graph_json":"https://pith.science/api/pith-number/3WYKEHWNJAKVYZ7EYVRMWYT7XM/graph.json","events_json":"https://pith.science/api/pith-number/3WYKEHWNJAKVYZ7EYVRMWYT7XM/events.json","paper":"https://pith.science/paper/3WYKEHWN"},"agent_actions":{"view_html":"https://pith.science/pith/3WYKEHWNJAKVYZ7EYVRMWYT7XM","download_json":"https://pith.science/pith/3WYKEHWNJAKVYZ7EYVRMWYT7XM.json","view_paper":"https://pith.science/paper/3WYKEHWN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.16838&json=true","fetch_graph":"https://pith.science/api/pith-number/3WYKEHWNJAKVYZ7EYVRMWYT7XM/graph.json","fetch_events":"https://pith.science/api/pith-number/3WYKEHWNJAKVYZ7EYVRMWYT7XM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3WYKEHWNJAKVYZ7EYVRMWYT7XM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3WYKEHWNJAKVYZ7EYVRMWYT7XM/action/storage_attestation","attest_author":"https://pith.science/pith/3WYKEHWNJAKVYZ7EYVRMWYT7XM/action/author_attestation","sign_citation":"https://pith.science/pith/3WYKEHWNJAKVYZ7EYVRMWYT7XM/action/citation_signature","submit_replication":"https://pith.science/pith/3WYKEHWNJAKVYZ7EYVRMWYT7XM/action/replication_record"}},"created_at":"2026-07-05T09:38:15.169863+00:00","updated_at":"2026-07-05T09:38:15.169863+00:00"}