{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LB4D5IBKV66EURPSGN4BTS3MJY","short_pith_number":"pith:LB4D5IBK","schema_version":"1.0","canonical_sha256":"58783ea02aafbc4a45f2337819cb6c4e25140009b60983f066dc4b4e0649e30b","source":{"kind":"arxiv","id":"2502.18878","version":2},"attestation_state":"computed","paper":{"title":"Learning to Generate Structured Output with Schema Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fangming Liu, Haolun Li, Maosong Sun, Xin Cong, Yankai Lin, Yaxi Lu, Yesai Wu, Zhiyuan Liu, Zhong Zhang","submitted_at":"2025-02-26T06:45:29Z","abstract_excerpt":"This study investigates the structured generation capabilities of large language models (LLMs), focusing on producing valid JSON outputs against a given schema. Despite the widespread use of JSON in integrating language models with programs, there is a lack of comprehensive analysis and benchmarking of these capabilities. We explore various aspects of JSON generation, such as structure understanding, escaping, and natural language description, to determine how to assess and enable LLMs to generate valid responses. Building upon this, we propose SchemaBench features around 40K different JSON sc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.18878","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-26T06:45:29Z","cross_cats_sorted":[],"title_canon_sha256":"2a86c00ebc2bc3717763f74aa726ce40e761483944ab34ef2381176f01b2bf72","abstract_canon_sha256":"ea43e2ad290722ddf994a0f7536ef15fe4fb4ef7635a7846ea7aff0f667c6d6f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:25:18.506451Z","signature_b64":"thsjMeeZsdNQvAA9YgknQosiD+iiXv8bcvQTlJdmNjtY/FHGtG8s53WDPZ/o0rG+SdOe6oRG7vd2b1AxE4hfAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"58783ea02aafbc4a45f2337819cb6c4e25140009b60983f066dc4b4e0649e30b","last_reissued_at":"2026-07-05T10:25:18.505967Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:25:18.505967Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning to Generate Structured Output with Schema Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fangming Liu, Haolun Li, Maosong Sun, Xin Cong, Yankai Lin, Yaxi Lu, Yesai Wu, Zhiyuan Liu, Zhong Zhang","submitted_at":"2025-02-26T06:45:29Z","abstract_excerpt":"This study investigates the structured generation capabilities of large language models (LLMs), focusing on producing valid JSON outputs against a given schema. Despite the widespread use of JSON in integrating language models with programs, there is a lack of comprehensive analysis and benchmarking of these capabilities. We explore various aspects of JSON generation, such as structure understanding, escaping, and natural language description, to determine how to assess and enable LLMs to generate valid responses. Building upon this, we propose SchemaBench features around 40K different JSON sc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.18878","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.18878/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.18878","created_at":"2026-07-05T10:25:18.506018+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.18878v2","created_at":"2026-07-05T10:25:18.506018+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.18878","created_at":"2026-07-05T10:25:18.506018+00:00"},{"alias_kind":"pith_short_12","alias_value":"LB4D5IBKV66E","created_at":"2026-07-05T10:25:18.506018+00:00"},{"alias_kind":"pith_short_16","alias_value":"LB4D5IBKV66EURPS","created_at":"2026-07-05T10:25:18.506018+00:00"},{"alias_kind":"pith_short_8","alias_value":"LB4D5IBK","created_at":"2026-07-05T10:25:18.506018+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08177","citing_title":"ASMR: Agentic Schema Generation for Ship Maintenance Report Writing","ref_index":11,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22817","citing_title":"SelPE: Progressive Selection for Private Structured Text Synthesis","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05206","citing_title":"Ontology-constrained multi-LLM scoring of hypothesis support in the predictive processing literature","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LB4D5IBKV66EURPSGN4BTS3MJY","json":"https://pith.science/pith/LB4D5IBKV66EURPSGN4BTS3MJY.json","graph_json":"https://pith.science/api/pith-number/LB4D5IBKV66EURPSGN4BTS3MJY/graph.json","events_json":"https://pith.science/api/pith-number/LB4D5IBKV66EURPSGN4BTS3MJY/events.json","paper":"https://pith.science/paper/LB4D5IBK"},"agent_actions":{"view_html":"https://pith.science/pith/LB4D5IBKV66EURPSGN4BTS3MJY","download_json":"https://pith.science/pith/LB4D5IBKV66EURPSGN4BTS3MJY.json","view_paper":"https://pith.science/paper/LB4D5IBK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.18878&json=true","fetch_graph":"https://pith.science/api/pith-number/LB4D5IBKV66EURPSGN4BTS3MJY/graph.json","fetch_events":"https://pith.science/api/pith-number/LB4D5IBKV66EURPSGN4BTS3MJY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LB4D5IBKV66EURPSGN4BTS3MJY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LB4D5IBKV66EURPSGN4BTS3MJY/action/storage_attestation","attest_author":"https://pith.science/pith/LB4D5IBKV66EURPSGN4BTS3MJY/action/author_attestation","sign_citation":"https://pith.science/pith/LB4D5IBKV66EURPSGN4BTS3MJY/action/citation_signature","submit_replication":"https://pith.science/pith/LB4D5IBKV66EURPSGN4BTS3MJY/action/replication_record"}},"created_at":"2026-07-05T10:25:18.506018+00:00","updated_at":"2026-07-05T10:25:18.506018+00:00"}