{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:D6TPSLJBDBTW4HRBZRHA4HQYJG","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"5daf950cecedc04a2a8e72e060e5c9198bf6ea4e7a78c259f5dbbce38bc22559","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-11-07T21:36:52Z","title_canon_sha256":"fb96ec099b640a0abe6b56efda25e195af9bce78b04304ec3d3d277f1326bf4d"},"schema_version":"1.0","source":{"id":"2411.05193","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2411.05193","created_at":"2026-07-05T09:41:03Z"},{"alias_kind":"arxiv_version","alias_value":"2411.05193v2","created_at":"2026-07-05T09:41:03Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.05193","created_at":"2026-07-05T09:41:03Z"},{"alias_kind":"pith_short_12","alias_value":"D6TPSLJBDBTW","created_at":"2026-07-05T09:41:03Z"},{"alias_kind":"pith_short_16","alias_value":"D6TPSLJBDBTW4HRB","created_at":"2026-07-05T09:41:03Z"},{"alias_kind":"pith_short_8","alias_value":"D6TPSLJB","created_at":"2026-07-05T09:41:03Z"}],"graph_snapshots":[{"event_id":"sha256:0eda0711a32b7fbbcfc62ad5c098216a655b23a7565164deafd68f418b44690c","target":"graph","created_at":"2026-07-05T09:41:03Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2411.05193/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Value-based reinforcement learning (RL) can in principle learn effective policies for a wide range of multi-turn problems, from games to dialogue to robotic control, including via offline RL from static previously collected datasets. However, despite the widespread use of policy gradient methods to train large language models for single turn tasks (e.g., question answering), value-based methods for multi-turn RL in an off-policy or offline setting have proven particularly challenging to scale to the setting of large language models. This setting requires effectively leveraging pretraining, sca","authors_text":"Anca Dragan, Joey Hong, Sergey Levine","cross_cats":["cs.AI","cs.CL"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-11-07T21:36:52Z","title":"Q-SFT: Q-Learning for Language Models via Supervised Fine-Tuning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.05193","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:ea99c3e29b4613be9ab1cff780c5476f01dd3f9fdf6b004342bcbeb1d42f15d8","target":"record","created_at":"2026-07-05T09:41:03Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"5daf950cecedc04a2a8e72e060e5c9198bf6ea4e7a78c259f5dbbce38bc22559","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-11-07T21:36:52Z","title_canon_sha256":"fb96ec099b640a0abe6b56efda25e195af9bce78b04304ec3d3d277f1326bf4d"},"schema_version":"1.0","source":{"id":"2411.05193","kind":"arxiv","version":2}},"canonical_sha256":"1fa6f92d2118676e1e21cc4e0e1e18499cf74dea77e3ee49387692fa03327176","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"1fa6f92d2118676e1e21cc4e0e1e18499cf74dea77e3ee49387692fa03327176","first_computed_at":"2026-07-05T09:41:03.417847Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:41:03.417847Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"HUVesIH8J8f7vc8tcd86c61S+ovAl4aAQJ70vOdLnN1EtuqCRbKBJQtGQcAopyWoqWp9mwA1X4dTgm12kUGYCw==","signature_status":"signed_v1","signed_at":"2026-07-05T09:41:03.418282Z","signed_message":"canonical_sha256_bytes"},"source_id":"2411.05193","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:ea99c3e29b4613be9ab1cff780c5476f01dd3f9fdf6b004342bcbeb1d42f15d8","sha256:0eda0711a32b7fbbcfc62ad5c098216a655b23a7565164deafd68f418b44690c"],"state_sha256":"bf85d5c8418686697a696b3c9f1c9c2eb5f8ee44b118203ff41c3cfe6bf7038e"}