{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EJ2HTZ5XS2LWHGZTQTXMPLQCWM","short_pith_number":"pith:EJ2HTZ5X","schema_version":"1.0","canonical_sha256":"227479e7b79697639b3384eec7ae02b32f2863055afe6ccdd3d941d073fe8220","source":{"kind":"arxiv","id":"2403.06988","version":1},"attestation_state":"computed","paper":{"title":"Guiding LLMs The Right Way: Fast, Non-Invasive Constrained Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Luca Beurer-Kellner, Marc Fischer, Martin Vechev","submitted_at":"2024-02-07T13:36:02Z","abstract_excerpt":"To ensure that text generated by large language models (LLMs) is in an expected format, constrained decoding proposes to enforce strict formal language constraints during generation. However, as we show in this work, not only do such methods incur performance overhead during generation, but many of them also significantly impair task accuracy, if they do not correctly align the underlying LLM sub-word vocabularies with external constraints. To address this, we present a novel decoding algorithm, DOMINO, that can enforce constraints in a fully subword-aligned fashion, while leveraging pre-compu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.06988","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-07T13:36:02Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"c2dfb0b64499a93987f8c9e6acc19421c1b1d1d6014068d244116b4043cbcbdc","abstract_canon_sha256":"494c2170e954845a90356c3c4d2f3dce77c3aa823a84e1edb38a2c21ed2750b2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:54:57.614589Z","signature_b64":"i9KNyRf1GiydZIXrkkJL9fUx7v84up6dwpNmjqVdF1HOWIYZwTLHNh64w7u6pbJ3mK2+2ej3KcxAJtrvVqBQBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"227479e7b79697639b3384eec7ae02b32f2863055afe6ccdd3d941d073fe8220","last_reissued_at":"2026-07-05T07:54:57.614113Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:54:57.614113Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Guiding LLMs The Right Way: Fast, Non-Invasive Constrained Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Luca Beurer-Kellner, Marc Fischer, Martin Vechev","submitted_at":"2024-02-07T13:36:02Z","abstract_excerpt":"To ensure that text generated by large language models (LLMs) is in an expected format, constrained decoding proposes to enforce strict formal language constraints during generation. However, as we show in this work, not only do such methods incur performance overhead during generation, but many of them also significantly impair task accuracy, if they do not correctly align the underlying LLM sub-word vocabularies with external constraints. To address this, we present a novel decoding algorithm, DOMINO, that can enforce constraints in a fully subword-aligned fashion, while leveraging pre-compu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.06988","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.06988/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.06988","created_at":"2026-07-05T07:54:57.614178+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.06988v1","created_at":"2026-07-05T07:54:57.614178+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.06988","created_at":"2026-07-05T07:54:57.614178+00:00"},{"alias_kind":"pith_short_12","alias_value":"EJ2HTZ5XS2LW","created_at":"2026-07-05T07:54:57.614178+00:00"},{"alias_kind":"pith_short_16","alias_value":"EJ2HTZ5XS2LWHGZT","created_at":"2026-07-05T07:54:57.614178+00:00"},{"alias_kind":"pith_short_8","alias_value":"EJ2HTZ5X","created_at":"2026-07-05T07:54:57.614178+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09395","citing_title":"Empirical Study for Structured Output Control in LLMs for Software Engineering","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13076","citing_title":"TruncProof: A Guardrail for LLM-based JSON Generation under Token-Length Constraints","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2506.23964","citing_title":"Making Logic a First-Class Citizen in Generative ML for Networking","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21086","citing_title":"Orthographic Constraint Satisfaction and Human Difficulty Alignment in Large Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13076","citing_title":"TruncProof: A Guardrail for LLM-based JSON Generation under Token-Length Constraints","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.28028","citing_title":"Reliable Answers for Recurring Questions: Boosting Text-to-SQL Accuracy with Template Constrained Decoding","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08737","citing_title":"The Extrapolation Cliff in On-Policy Distillation of Near-Deterministic Structured Outputs","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10318","citing_title":"Extending Confidence-Based Text2Cypher with Grammar and Schema Aware Filtering","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10513","citing_title":"Agent Mentor: Framing Agent Knowledge through Semantic Trajectory Analysis","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10114","citing_title":"CircuitSynth: Reliable Synthetic Data Generation","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EJ2HTZ5XS2LWHGZTQTXMPLQCWM","json":"https://pith.science/pith/EJ2HTZ5XS2LWHGZTQTXMPLQCWM.json","graph_json":"https://pith.science/api/pith-number/EJ2HTZ5XS2LWHGZTQTXMPLQCWM/graph.json","events_json":"https://pith.science/api/pith-number/EJ2HTZ5XS2LWHGZTQTXMPLQCWM/events.json","paper":"https://pith.science/paper/EJ2HTZ5X"},"agent_actions":{"view_html":"https://pith.science/pith/EJ2HTZ5XS2LWHGZTQTXMPLQCWM","download_json":"https://pith.science/pith/EJ2HTZ5XS2LWHGZTQTXMPLQCWM.json","view_paper":"https://pith.science/paper/EJ2HTZ5X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.06988&json=true","fetch_graph":"https://pith.science/api/pith-number/EJ2HTZ5XS2LWHGZTQTXMPLQCWM/graph.json","fetch_events":"https://pith.science/api/pith-number/EJ2HTZ5XS2LWHGZTQTXMPLQCWM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EJ2HTZ5XS2LWHGZTQTXMPLQCWM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EJ2HTZ5XS2LWHGZTQTXMPLQCWM/action/storage_attestation","attest_author":"https://pith.science/pith/EJ2HTZ5XS2LWHGZTQTXMPLQCWM/action/author_attestation","sign_citation":"https://pith.science/pith/EJ2HTZ5XS2LWHGZTQTXMPLQCWM/action/citation_signature","submit_replication":"https://pith.science/pith/EJ2HTZ5XS2LWHGZTQTXMPLQCWM/action/replication_record"}},"created_at":"2026-07-05T07:54:57.614178+00:00","updated_at":"2026-07-05T07:54:57.614178+00:00"}