{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:O7GZ4G57JC3I5L37ASZCWCCYO4","short_pith_number":"pith:O7GZ4G57","schema_version":"1.0","canonical_sha256":"77cd9e1bbf48b68eaf7f04b22b085877384b70b19a7eca1c394616f897c86e48","source":{"kind":"arxiv","id":"2405.11422","version":1},"attestation_state":"computed","paper":{"title":"Large Language Models are Biased Reinforcement Learners","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Nicolas Yax, Stefano Palminteri, William M. Hayes","submitted_at":"2024-05-19T01:43:52Z","abstract_excerpt":"In-context learning enables large language models (LLMs) to perform a variety of tasks, including learning to make reward-maximizing choices in simple bandit tasks. Given their potential use as (autonomous) decision-making agents, it is important to understand how these models perform such reinforcement learning (RL) tasks and the extent to which they are susceptible to biases. Motivated by the fact that, in humans, it has been widely documented that the value of an outcome depends on how it compares to other local outcomes, the present study focuses on whether similar value encoding biases ap"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.11422","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-19T01:43:52Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"5852554dce3c6466ed02a9305228a23e0bbbd2aea58f59210cb38ebe2bedbfd3","abstract_canon_sha256":"09cfeccf29d2fd2805e573f58f71bf7c21eeb3d102c7b01be20ec2019b6e3b32"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:20:39.148483Z","signature_b64":"lLxJb9tOePOGSOxQqvQaYE71HPZ3s5sd8yH5gQlBwm+k76s88wWRLkCPuVBX1iLVOECp5U2hDg8UrvBe6Bn2AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"77cd9e1bbf48b68eaf7f04b22b085877384b70b19a7eca1c394616f897c86e48","last_reissued_at":"2026-07-05T08:20:39.148094Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:20:39.148094Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Models are Biased Reinforcement Learners","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Nicolas Yax, Stefano Palminteri, William M. Hayes","submitted_at":"2024-05-19T01:43:52Z","abstract_excerpt":"In-context learning enables large language models (LLMs) to perform a variety of tasks, including learning to make reward-maximizing choices in simple bandit tasks. Given their potential use as (autonomous) decision-making agents, it is important to understand how these models perform such reinforcement learning (RL) tasks and the extent to which they are susceptible to biases. Motivated by the fact that, in humans, it has been widely documented that the value of an outcome depends on how it compares to other local outcomes, the present study focuses on whether similar value encoding biases ap"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.11422","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.11422/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.11422","created_at":"2026-07-05T08:20:39.148153+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.11422v1","created_at":"2026-07-05T08:20:39.148153+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.11422","created_at":"2026-07-05T08:20:39.148153+00:00"},{"alias_kind":"pith_short_12","alias_value":"O7GZ4G57JC3I","created_at":"2026-07-05T08:20:39.148153+00:00"},{"alias_kind":"pith_short_16","alias_value":"O7GZ4G57JC3I5L37","created_at":"2026-07-05T08:20:39.148153+00:00"},{"alias_kind":"pith_short_8","alias_value":"O7GZ4G57","created_at":"2026-07-05T08:20:39.148153+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07988","citing_title":"PAFO: Pareto Fairness Optimization for Personalized Reward Modeling","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O7GZ4G57JC3I5L37ASZCWCCYO4","json":"https://pith.science/pith/O7GZ4G57JC3I5L37ASZCWCCYO4.json","graph_json":"https://pith.science/api/pith-number/O7GZ4G57JC3I5L37ASZCWCCYO4/graph.json","events_json":"https://pith.science/api/pith-number/O7GZ4G57JC3I5L37ASZCWCCYO4/events.json","paper":"https://pith.science/paper/O7GZ4G57"},"agent_actions":{"view_html":"https://pith.science/pith/O7GZ4G57JC3I5L37ASZCWCCYO4","download_json":"https://pith.science/pith/O7GZ4G57JC3I5L37ASZCWCCYO4.json","view_paper":"https://pith.science/paper/O7GZ4G57","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.11422&json=true","fetch_graph":"https://pith.science/api/pith-number/O7GZ4G57JC3I5L37ASZCWCCYO4/graph.json","fetch_events":"https://pith.science/api/pith-number/O7GZ4G57JC3I5L37ASZCWCCYO4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O7GZ4G57JC3I5L37ASZCWCCYO4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O7GZ4G57JC3I5L37ASZCWCCYO4/action/storage_attestation","attest_author":"https://pith.science/pith/O7GZ4G57JC3I5L37ASZCWCCYO4/action/author_attestation","sign_citation":"https://pith.science/pith/O7GZ4G57JC3I5L37ASZCWCCYO4/action/citation_signature","submit_replication":"https://pith.science/pith/O7GZ4G57JC3I5L37ASZCWCCYO4/action/replication_record"}},"created_at":"2026-07-05T08:20:39.148153+00:00","updated_at":"2026-07-05T08:20:39.148153+00:00"}