{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SZHZPKIBEVW2RT4UC27TL27F5V","short_pith_number":"pith:SZHZPKIB","schema_version":"1.0","canonical_sha256":"964f97a901256da8cf9416bf35ebe5ed5cb516b67517a8ebc6d604ea00525975","source":{"kind":"arxiv","id":"2403.11558","version":1},"attestation_state":"computed","paper":{"title":"Reinforcement Learning with Token-level Feedback for Controllable Text Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dangyang Chen, Kaihe Xu, Wei Wei, Wendi Li, Wenfeng Xie, Yu Cheng","submitted_at":"2024-03-18T08:18:37Z","abstract_excerpt":"To meet the requirements of real-world applications, it is essential to control generations of large language models (LLMs). Prior research has tried to introduce reinforcement learning (RL) into controllable text generation while most existing methods suffer from overfitting issues (finetuning-based methods) or semantic collapse (post-processing methods). However, current RL methods are generally guided by coarse-grained (sentence/paragraph-level) feedback, which may lead to suboptimal performance owing to semantic twists or progressions within sentences. To tackle that, we propose a novel re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.11558","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-03-18T08:18:37Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"220f6248e5419ca59d57d10f2701b8d877f1a9798a9af2f6c681517667fb71b6","abstract_canon_sha256":"b5dedc1f1f37071c5f2a7300da2bfbfc13a41202d2a923219e037d8870565309"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:57:18.325567Z","signature_b64":"viKZmYCVTZ17HBXTbUgNKKq8oPpy17K6WS++v9SjtA/lQTNsCYQXRIDhHVXE9T5DUE1yQHwkD7U83DVrSdBeCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"964f97a901256da8cf9416bf35ebe5ed5cb516b67517a8ebc6d604ea00525975","last_reissued_at":"2026-07-05T07:57:18.325013Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:57:18.325013Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reinforcement Learning with Token-level Feedback for Controllable Text Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dangyang Chen, Kaihe Xu, Wei Wei, Wendi Li, Wenfeng Xie, Yu Cheng","submitted_at":"2024-03-18T08:18:37Z","abstract_excerpt":"To meet the requirements of real-world applications, it is essential to control generations of large language models (LLMs). Prior research has tried to introduce reinforcement learning (RL) into controllable text generation while most existing methods suffer from overfitting issues (finetuning-based methods) or semantic collapse (post-processing methods). However, current RL methods are generally guided by coarse-grained (sentence/paragraph-level) feedback, which may lead to suboptimal performance owing to semantic twists or progressions within sentences. To tackle that, we propose a novel re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.11558","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.11558/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.11558","created_at":"2026-07-05T07:57:18.325090+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.11558v1","created_at":"2026-07-05T07:57:18.325090+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.11558","created_at":"2026-07-05T07:57:18.325090+00:00"},{"alias_kind":"pith_short_12","alias_value":"SZHZPKIBEVW2","created_at":"2026-07-05T07:57:18.325090+00:00"},{"alias_kind":"pith_short_16","alias_value":"SZHZPKIBEVW2RT4U","created_at":"2026-07-05T07:57:18.325090+00:00"},{"alias_kind":"pith_short_8","alias_value":"SZHZPKIB","created_at":"2026-07-05T07:57:18.325090+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SZHZPKIBEVW2RT4UC27TL27F5V","json":"https://pith.science/pith/SZHZPKIBEVW2RT4UC27TL27F5V.json","graph_json":"https://pith.science/api/pith-number/SZHZPKIBEVW2RT4UC27TL27F5V/graph.json","events_json":"https://pith.science/api/pith-number/SZHZPKIBEVW2RT4UC27TL27F5V/events.json","paper":"https://pith.science/paper/SZHZPKIB"},"agent_actions":{"view_html":"https://pith.science/pith/SZHZPKIBEVW2RT4UC27TL27F5V","download_json":"https://pith.science/pith/SZHZPKIBEVW2RT4UC27TL27F5V.json","view_paper":"https://pith.science/paper/SZHZPKIB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.11558&json=true","fetch_graph":"https://pith.science/api/pith-number/SZHZPKIBEVW2RT4UC27TL27F5V/graph.json","fetch_events":"https://pith.science/api/pith-number/SZHZPKIBEVW2RT4UC27TL27F5V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SZHZPKIBEVW2RT4UC27TL27F5V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SZHZPKIBEVW2RT4UC27TL27F5V/action/storage_attestation","attest_author":"https://pith.science/pith/SZHZPKIBEVW2RT4UC27TL27F5V/action/author_attestation","sign_citation":"https://pith.science/pith/SZHZPKIBEVW2RT4UC27TL27F5V/action/citation_signature","submit_replication":"https://pith.science/pith/SZHZPKIBEVW2RT4UC27TL27F5V/action/replication_record"}},"created_at":"2026-07-05T07:57:18.325090+00:00","updated_at":"2026-07-05T07:57:18.325090+00:00"}