{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:B7T6I6XL4XT43BD7FJX2TXEJF6","short_pith_number":"pith:B7T6I6XL","schema_version":"1.0","canonical_sha256":"0fe7e47aebe5e7cd847f2a6fa9dc892f96f996f3e78320bb3df5620eb96cbec5","source":{"kind":"arxiv","id":"2505.23927","version":1},"attestation_state":"computed","paper":{"title":"Thompson Sampling in Online RLHF with General Function Approximation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jie Fu, Songtao Feng","submitted_at":"2025-05-29T18:22:02Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) has achieved great empirical success in aligning large language models (LLMs) with human preference, and it is of great importance to study the statistical efficiency of RLHF algorithms from a theoretical perspective. In this work, we consider the online RLHF setting where the preference data is revealed during the learning process and study action value function approximation. We design a model-free posterior sampling algorithm for online RLHF inspired by Thompson sampling and provide its theoretical guarantee. Specifically, we adopt Bellman e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.23927","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-29T18:22:02Z","cross_cats_sorted":[],"title_canon_sha256":"84645f304f59984b80befa9139272fa1a7a63327a2623482a059795dd760a477","abstract_canon_sha256":"2cfb135eeb8fa37b38a0133397b0f181dee8b56f43268ed9e7422b1f7b8ad8e9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:37.946984Z","signature_b64":"Don/nHh+nRaVUyCUN1rJ9EPoLRu/yV0vSCzYRM/C1AuGL6JpfhbnOqctT4yZ4HdXf7TVN1qxVKWgsbKQL0FkBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0fe7e47aebe5e7cd847f2a6fa9dc892f96f996f3e78320bb3df5620eb96cbec5","last_reissued_at":"2026-07-05T11:12:37.946486Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:37.946486Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Thompson Sampling in Online RLHF with General Function Approximation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jie Fu, Songtao Feng","submitted_at":"2025-05-29T18:22:02Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) has achieved great empirical success in aligning large language models (LLMs) with human preference, and it is of great importance to study the statistical efficiency of RLHF algorithms from a theoretical perspective. In this work, we consider the online RLHF setting where the preference data is revealed during the learning process and study action value function approximation. We design a model-free posterior sampling algorithm for online RLHF inspired by Thompson sampling and provide its theoretical guarantee. Specifically, we adopt Bellman e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23927","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.23927/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.23927","created_at":"2026-07-05T11:12:37.946547+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.23927v1","created_at":"2026-07-05T11:12:37.946547+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23927","created_at":"2026-07-05T11:12:37.946547+00:00"},{"alias_kind":"pith_short_12","alias_value":"B7T6I6XL4XT4","created_at":"2026-07-05T11:12:37.946547+00:00"},{"alias_kind":"pith_short_16","alias_value":"B7T6I6XL4XT43BD7","created_at":"2026-07-05T11:12:37.946547+00:00"},{"alias_kind":"pith_short_8","alias_value":"B7T6I6XL","created_at":"2026-07-05T11:12:37.946547+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B7T6I6XL4XT43BD7FJX2TXEJF6","json":"https://pith.science/pith/B7T6I6XL4XT43BD7FJX2TXEJF6.json","graph_json":"https://pith.science/api/pith-number/B7T6I6XL4XT43BD7FJX2TXEJF6/graph.json","events_json":"https://pith.science/api/pith-number/B7T6I6XL4XT43BD7FJX2TXEJF6/events.json","paper":"https://pith.science/paper/B7T6I6XL"},"agent_actions":{"view_html":"https://pith.science/pith/B7T6I6XL4XT43BD7FJX2TXEJF6","download_json":"https://pith.science/pith/B7T6I6XL4XT43BD7FJX2TXEJF6.json","view_paper":"https://pith.science/paper/B7T6I6XL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.23927&json=true","fetch_graph":"https://pith.science/api/pith-number/B7T6I6XL4XT43BD7FJX2TXEJF6/graph.json","fetch_events":"https://pith.science/api/pith-number/B7T6I6XL4XT43BD7FJX2TXEJF6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B7T6I6XL4XT43BD7FJX2TXEJF6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B7T6I6XL4XT43BD7FJX2TXEJF6/action/storage_attestation","attest_author":"https://pith.science/pith/B7T6I6XL4XT43BD7FJX2TXEJF6/action/author_attestation","sign_citation":"https://pith.science/pith/B7T6I6XL4XT43BD7FJX2TXEJF6/action/citation_signature","submit_replication":"https://pith.science/pith/B7T6I6XL4XT43BD7FJX2TXEJF6/action/replication_record"}},"created_at":"2026-07-05T11:12:37.946547+00:00","updated_at":"2026-07-05T11:12:37.946547+00:00"}