{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ILL55OTU5AUSMYLBYFYB2PNJXO","short_pith_number":"pith:ILL55OTU","schema_version":"1.0","canonical_sha256":"42d7deba74e829266161c1701d3da9bbadb092893dfe3b7714a8cb2238515f2d","source":{"kind":"arxiv","id":"2407.08103","version":3},"attestation_state":"computed","paper":{"title":"Automata-based constraints for language model decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.FL"],"primary_cat":"cs.CL","authors_text":"Frederick Liu, Luheng He, Terry Koo","submitted_at":"2024-07-11T00:25:01Z","abstract_excerpt":"Language models (LMs) are often expected to generate strings in some formal language; for example, structured data, API calls, or code snippets. Although LMs can be tuned to improve their adherence to formal syntax, this does not guarantee conformance, especially with smaller LMs suitable for large-scale deployment. In addition, tuning requires significant resources, making it impractical for uncommon or task-specific formats. To prevent downstream parsing errors we would ideally constrain the LM to only produce valid output, but this is severely complicated by tokenization, which is typically"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.08103","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-11T00:25:01Z","cross_cats_sorted":["cs.FL"],"title_canon_sha256":"70ec17a348da4c8a4d5a81e9ee451ee9cd3bd0d73bf403d6c38b9941aba5ea0a","abstract_canon_sha256":"279724a4b1a5c49463106c8a939564524a09d67bea33d78a2ef95f827f458032"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:51:54.563620Z","signature_b64":"jC6596aFRF1dFrNfyJ5q8Gs+9VkAYMw43j/0oEfqxocejjWBuxLZQdSv9Im+39K6pWbrp6/OAOAiL6CXVnrsCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"42d7deba74e829266161c1701d3da9bbadb092893dfe3b7714a8cb2238515f2d","last_reissued_at":"2026-07-05T08:51:54.563108Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:51:54.563108Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Automata-based constraints for language model decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.FL"],"primary_cat":"cs.CL","authors_text":"Frederick Liu, Luheng He, Terry Koo","submitted_at":"2024-07-11T00:25:01Z","abstract_excerpt":"Language models (LMs) are often expected to generate strings in some formal language; for example, structured data, API calls, or code snippets. Although LMs can be tuned to improve their adherence to formal syntax, this does not guarantee conformance, especially with smaller LMs suitable for large-scale deployment. In addition, tuning requires significant resources, making it impractical for uncommon or task-specific formats. To prevent downstream parsing errors we would ideally constrain the LM to only produce valid output, but this is severely complicated by tokenization, which is typically"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.08103","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.08103/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.08103","created_at":"2026-07-05T08:51:54.563170+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.08103v3","created_at":"2026-07-05T08:51:54.563170+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.08103","created_at":"2026-07-05T08:51:54.563170+00:00"},{"alias_kind":"pith_short_12","alias_value":"ILL55OTU5AUS","created_at":"2026-07-05T08:51:54.563170+00:00"},{"alias_kind":"pith_short_16","alias_value":"ILL55OTU5AUSMYLB","created_at":"2026-07-05T08:51:54.563170+00:00"},{"alias_kind":"pith_short_8","alias_value":"ILL55OTU","created_at":"2026-07-05T08:51:54.563170+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11203","citing_title":"LatticeBridge: Rare-Event Sequential Inference for Faithful Structured Sequence Synthesis","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.14122","citing_title":"Beyond Perplexity: UTF-8 Validity in Byte-aware Language Models","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ILL55OTU5AUSMYLBYFYB2PNJXO","json":"https://pith.science/pith/ILL55OTU5AUSMYLBYFYB2PNJXO.json","graph_json":"https://pith.science/api/pith-number/ILL55OTU5AUSMYLBYFYB2PNJXO/graph.json","events_json":"https://pith.science/api/pith-number/ILL55OTU5AUSMYLBYFYB2PNJXO/events.json","paper":"https://pith.science/paper/ILL55OTU"},"agent_actions":{"view_html":"https://pith.science/pith/ILL55OTU5AUSMYLBYFYB2PNJXO","download_json":"https://pith.science/pith/ILL55OTU5AUSMYLBYFYB2PNJXO.json","view_paper":"https://pith.science/paper/ILL55OTU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.08103&json=true","fetch_graph":"https://pith.science/api/pith-number/ILL55OTU5AUSMYLBYFYB2PNJXO/graph.json","fetch_events":"https://pith.science/api/pith-number/ILL55OTU5AUSMYLBYFYB2PNJXO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ILL55OTU5AUSMYLBYFYB2PNJXO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ILL55OTU5AUSMYLBYFYB2PNJXO/action/storage_attestation","attest_author":"https://pith.science/pith/ILL55OTU5AUSMYLBYFYB2PNJXO/action/author_attestation","sign_citation":"https://pith.science/pith/ILL55OTU5AUSMYLBYFYB2PNJXO/action/citation_signature","submit_replication":"https://pith.science/pith/ILL55OTU5AUSMYLBYFYB2PNJXO/action/replication_record"}},"created_at":"2026-07-05T08:51:54.563170+00:00","updated_at":"2026-07-05T08:51:54.563170+00:00"}