{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:L35O5XZNRQX27XPGYRG3XFSU25","short_pith_number":"pith:L35O5XZN","schema_version":"1.0","canonical_sha256":"5efaeedf2d8c2fafdde6c44dbb9654d77006511faa231d4ee83ab156fcb93909","source":{"kind":"arxiv","id":"2410.02504","version":2},"attestation_state":"computed","paper":{"title":"Dual Active Learning for Reinforcement Learning from Human Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Chengchun Shi, Pangpang Liu, Will Wei Sun","submitted_at":"2024-10-03T14:09:58Z","abstract_excerpt":"Aligning large language models (LLMs) with human preferences is critical to recent advances in generative artificial intelligence. Reinforcement learning from human feedback (RLHF) is widely applied to achieve this objective. A key step in RLHF is to learn the reward function from human feedback. However, human feedback is costly and time-consuming, making it essential to collect high-quality conversation data for human teachers to label. Additionally, different human teachers have different levels of expertise. It is thus critical to query the most appropriate teacher for their opinions. In t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.02504","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2024-10-03T14:09:58Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"596627a9f5029e1a483ff50cbfaa8023bd02a50a694a05bf09636f77ff18738d","abstract_canon_sha256":"e36326ce0e10a31931f2290983b980c5511bff49166f63ae6cd3aacbe554fe95"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:55:34.988602Z","signature_b64":"XYBjel0MoqMLKJhkRDhVi0NQ+xvRmoeIKqhNbmHA+f7UUmdsPNYqGASWSYdvZtoPIPeaop+jlCqT1qfXiC9gDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5efaeedf2d8c2fafdde6c44dbb9654d77006511faa231d4ee83ab156fcb93909","last_reissued_at":"2026-07-05T09:55:34.988059Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:55:34.988059Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dual Active Learning for Reinforcement Learning from Human Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Chengchun Shi, Pangpang Liu, Will Wei Sun","submitted_at":"2024-10-03T14:09:58Z","abstract_excerpt":"Aligning large language models (LLMs) with human preferences is critical to recent advances in generative artificial intelligence. Reinforcement learning from human feedback (RLHF) is widely applied to achieve this objective. A key step in RLHF is to learn the reward function from human feedback. However, human feedback is costly and time-consuming, making it essential to collect high-quality conversation data for human teachers to label. Additionally, different human teachers have different levels of expertise. It is thus critical to query the most appropriate teacher for their opinions. In t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.02504","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.02504/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.02504","created_at":"2026-07-05T09:55:34.988125+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.02504v2","created_at":"2026-07-05T09:55:34.988125+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.02504","created_at":"2026-07-05T09:55:34.988125+00:00"},{"alias_kind":"pith_short_12","alias_value":"L35O5XZNRQX2","created_at":"2026-07-05T09:55:34.988125+00:00"},{"alias_kind":"pith_short_16","alias_value":"L35O5XZNRQX27XPG","created_at":"2026-07-05T09:55:34.988125+00:00"},{"alias_kind":"pith_short_8","alias_value":"L35O5XZN","created_at":"2026-07-05T09:55:34.988125+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24331","citing_title":"CurveRL: Principled Distribution-Aware Context Reweighting for LLM Reasoning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25864","citing_title":"When Self-Belief Misleads: Active Label Acquisition for Reinforcement Learning with Verifiable Rewards","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13284","citing_title":"Learning Perturbations to Extrapolate Your LLM","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02507","citing_title":"Reinforcement Learning from Human Feedback: A Statistical Perspective","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10784","citing_title":"MASS-DPO: Multi-negative Active Sample Selection for Direct Policy Optimization","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04344","citing_title":"Perturbation is All You Need for Extrapolating Language Models","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L35O5XZNRQX27XPGYRG3XFSU25","json":"https://pith.science/pith/L35O5XZNRQX27XPGYRG3XFSU25.json","graph_json":"https://pith.science/api/pith-number/L35O5XZNRQX27XPGYRG3XFSU25/graph.json","events_json":"https://pith.science/api/pith-number/L35O5XZNRQX27XPGYRG3XFSU25/events.json","paper":"https://pith.science/paper/L35O5XZN"},"agent_actions":{"view_html":"https://pith.science/pith/L35O5XZNRQX27XPGYRG3XFSU25","download_json":"https://pith.science/pith/L35O5XZNRQX27XPGYRG3XFSU25.json","view_paper":"https://pith.science/paper/L35O5XZN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.02504&json=true","fetch_graph":"https://pith.science/api/pith-number/L35O5XZNRQX27XPGYRG3XFSU25/graph.json","fetch_events":"https://pith.science/api/pith-number/L35O5XZNRQX27XPGYRG3XFSU25/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L35O5XZNRQX27XPGYRG3XFSU25/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L35O5XZNRQX27XPGYRG3XFSU25/action/storage_attestation","attest_author":"https://pith.science/pith/L35O5XZNRQX27XPGYRG3XFSU25/action/author_attestation","sign_citation":"https://pith.science/pith/L35O5XZNRQX27XPGYRG3XFSU25/action/citation_signature","submit_replication":"https://pith.science/pith/L35O5XZNRQX27XPGYRG3XFSU25/action/replication_record"}},"created_at":"2026-07-05T09:55:34.988125+00:00","updated_at":"2026-07-05T09:55:34.988125+00:00"}