{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PCBQHV4JPFBKI3I7ALVLXDOINX","short_pith_number":"pith:PCBQHV4J","schema_version":"1.0","canonical_sha256":"788303d7897942a46d1f02eabb8dc86dd4782c34867b8af0bcc83f4965e3b897","source":{"kind":"arxiv","id":"2406.11191","version":2},"attestation_state":"computed","paper":{"title":"A Survey on Human Preference Learning for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Juntao Li, Kehai Chen, Liqiang Nie, Min Zhang, Muyun Yang, Ruili Jiang, Tiejun Zhao, Xuefeng Bai, Zhixuan He","submitted_at":"2024-06-17T03:52:51Z","abstract_excerpt":"The recent surge of versatile large language models (LLMs) largely depends on aligning increasingly capable foundation models with human intentions by preference learning, enhancing LLMs with excellent applicability and effectiveness in a wide range of contexts. Despite the numerous related studies conducted, a perspective on how human preferences are introduced into LLMs remains limited, which may prevent a deeper comprehension of the relationships between human preferences and LLMs as well as the realization of their limitations. In this survey, we review the progress in exploring human pref"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.11191","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-17T03:52:51Z","cross_cats_sorted":[],"title_canon_sha256":"a8625b3f66e6258a36c2b4d42f5c4a5f781280a633c3dc2d68024262112d7422","abstract_canon_sha256":"eb61dd12ca4c342a11665f0e6e393f1393de1c4c658dd697eb3502dd398faf38"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:33:41.575332Z","signature_b64":"M1ee3tCynJkiqMlyKJK9qVqF1CO2u8Vac6b6QJ5NqG3CHbn9WGhA/c9hGyKaZb75jKlWVwACE/gJwZOaXw4rDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"788303d7897942a46d1f02eabb8dc86dd4782c34867b8af0bcc83f4965e3b897","last_reissued_at":"2026-07-05T08:33:41.574854Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:33:41.574854Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey on Human Preference Learning for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Juntao Li, Kehai Chen, Liqiang Nie, Min Zhang, Muyun Yang, Ruili Jiang, Tiejun Zhao, Xuefeng Bai, Zhixuan He","submitted_at":"2024-06-17T03:52:51Z","abstract_excerpt":"The recent surge of versatile large language models (LLMs) largely depends on aligning increasingly capable foundation models with human intentions by preference learning, enhancing LLMs with excellent applicability and effectiveness in a wide range of contexts. Despite the numerous related studies conducted, a perspective on how human preferences are introduced into LLMs remains limited, which may prevent a deeper comprehension of the relationships between human preferences and LLMs as well as the realization of their limitations. In this survey, we review the progress in exploring human pref"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.11191","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.11191/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.11191","created_at":"2026-07-05T08:33:41.574919+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.11191v2","created_at":"2026-07-05T08:33:41.574919+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.11191","created_at":"2026-07-05T08:33:41.574919+00:00"},{"alias_kind":"pith_short_12","alias_value":"PCBQHV4JPFBK","created_at":"2026-07-05T08:33:41.574919+00:00"},{"alias_kind":"pith_short_16","alias_value":"PCBQHV4JPFBKI3I7","created_at":"2026-07-05T08:33:41.574919+00:00"},{"alias_kind":"pith_short_8","alias_value":"PCBQHV4J","created_at":"2026-07-05T08:33:41.574919+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08274","citing_title":"Toward Human-Centered Multi-Agent Systems: Integrating Cognition, Culture, Values, and Cooperation in AI Agents","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28707","citing_title":"BV-Blend: Uncertainty-Weighted Historical Baselines for Stable Critic-Free RL with Verifiable Rewards","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2508.16771","citing_title":"EyeMulator: Improving Code Language Models by Mimicking Human Visual Attention","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2511.20284","citing_title":"Can LLMs Make (Personalized) Access Control Decisions?","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05593","citing_title":"Label Effects: Shared Heuristic Reliance in Trust Assessment by Humans and LLM-as-a-Judge","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PCBQHV4JPFBKI3I7ALVLXDOINX","json":"https://pith.science/pith/PCBQHV4JPFBKI3I7ALVLXDOINX.json","graph_json":"https://pith.science/api/pith-number/PCBQHV4JPFBKI3I7ALVLXDOINX/graph.json","events_json":"https://pith.science/api/pith-number/PCBQHV4JPFBKI3I7ALVLXDOINX/events.json","paper":"https://pith.science/paper/PCBQHV4J"},"agent_actions":{"view_html":"https://pith.science/pith/PCBQHV4JPFBKI3I7ALVLXDOINX","download_json":"https://pith.science/pith/PCBQHV4JPFBKI3I7ALVLXDOINX.json","view_paper":"https://pith.science/paper/PCBQHV4J","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.11191&json=true","fetch_graph":"https://pith.science/api/pith-number/PCBQHV4JPFBKI3I7ALVLXDOINX/graph.json","fetch_events":"https://pith.science/api/pith-number/PCBQHV4JPFBKI3I7ALVLXDOINX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PCBQHV4JPFBKI3I7ALVLXDOINX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PCBQHV4JPFBKI3I7ALVLXDOINX/action/storage_attestation","attest_author":"https://pith.science/pith/PCBQHV4JPFBKI3I7ALVLXDOINX/action/author_attestation","sign_citation":"https://pith.science/pith/PCBQHV4JPFBKI3I7ALVLXDOINX/action/citation_signature","submit_replication":"https://pith.science/pith/PCBQHV4JPFBKI3I7ALVLXDOINX/action/replication_record"}},"created_at":"2026-07-05T08:33:41.574919+00:00","updated_at":"2026-07-05T08:33:41.574919+00:00"}