{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:MNP4IEIFGKLPEGVN4BLIUGTS6B","short_pith_number":"pith:MNP4IEIF","schema_version":"1.0","canonical_sha256":"635fc411053296f21aade0568a1a72f06b7e9d1f987c0fb9e8928e2831869bef","source":{"kind":"arxiv","id":"2310.15017","version":3},"attestation_state":"computed","paper":{"title":"Mind the Model, Not the Agent: The Primacy Bias in Model-based RL","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jiafei Lyu, Xiu Li, Zhongjian Qiao","submitted_at":"2023-10-23T15:12:20Z","abstract_excerpt":"The primacy bias in model-free reinforcement learning (MFRL), which refers to the agent's tendency to overfit early data and lose the ability to learn from new data, can significantly decrease the performance of MFRL algorithms. Previous studies have shown that employing simple techniques, such as resetting the agent's parameters, can substantially alleviate the primacy bias in MFRL. However, the primacy bias in model-based reinforcement learning (MBRL) remains unexplored. In this work, we focus on investigating the primacy bias in MBRL. We begin by observing that resetting the agent's paramet"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.15017","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-10-23T15:12:20Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2a4e8c032a813f5a55eaa31c6764306567e70e344a7b0939ed45811437399ebf","abstract_canon_sha256":"1beb2ad996e539a72d266d078050935a68a5dc0e2b50ebf4ce52a57fdd5f2e0c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:56:14.746235Z","signature_b64":"SLtp0lNK9oYI83TUaJR0OjgKRxPQcMZX31GPALGBZCmQMU6D/1hBKk3czkCLLymV4lmRFhgQ58YRswW27yT0Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"635fc411053296f21aade0568a1a72f06b7e9d1f987c0fb9e8928e2831869bef","last_reissued_at":"2026-07-05T08:56:14.745815Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:56:14.745815Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mind the Model, Not the Agent: The Primacy Bias in Model-based RL","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jiafei Lyu, Xiu Li, Zhongjian Qiao","submitted_at":"2023-10-23T15:12:20Z","abstract_excerpt":"The primacy bias in model-free reinforcement learning (MFRL), which refers to the agent's tendency to overfit early data and lose the ability to learn from new data, can significantly decrease the performance of MFRL algorithms. Previous studies have shown that employing simple techniques, such as resetting the agent's parameters, can substantially alleviate the primacy bias in MFRL. However, the primacy bias in model-based reinforcement learning (MBRL) remains unexplored. In this work, we focus on investigating the primacy bias in MBRL. We begin by observing that resetting the agent's paramet"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.15017","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.15017/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.15017","created_at":"2026-07-05T08:56:14.745869+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.15017v3","created_at":"2026-07-05T08:56:14.745869+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.15017","created_at":"2026-07-05T08:56:14.745869+00:00"},{"alias_kind":"pith_short_12","alias_value":"MNP4IEIFGKLP","created_at":"2026-07-05T08:56:14.745869+00:00"},{"alias_kind":"pith_short_16","alias_value":"MNP4IEIFGKLPEGVN","created_at":"2026-07-05T08:56:14.745869+00:00"},{"alias_kind":"pith_short_8","alias_value":"MNP4IEIF","created_at":"2026-07-05T08:56:14.745869+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.02712","citing_title":"A Forget-and-Grow Strategy for Deep Reinforcement Learning Scaling in Continuous Control","ref_index":39,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MNP4IEIFGKLPEGVN4BLIUGTS6B","json":"https://pith.science/pith/MNP4IEIFGKLPEGVN4BLIUGTS6B.json","graph_json":"https://pith.science/api/pith-number/MNP4IEIFGKLPEGVN4BLIUGTS6B/graph.json","events_json":"https://pith.science/api/pith-number/MNP4IEIFGKLPEGVN4BLIUGTS6B/events.json","paper":"https://pith.science/paper/MNP4IEIF"},"agent_actions":{"view_html":"https://pith.science/pith/MNP4IEIFGKLPEGVN4BLIUGTS6B","download_json":"https://pith.science/pith/MNP4IEIFGKLPEGVN4BLIUGTS6B.json","view_paper":"https://pith.science/paper/MNP4IEIF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.15017&json=true","fetch_graph":"https://pith.science/api/pith-number/MNP4IEIFGKLPEGVN4BLIUGTS6B/graph.json","fetch_events":"https://pith.science/api/pith-number/MNP4IEIFGKLPEGVN4BLIUGTS6B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MNP4IEIFGKLPEGVN4BLIUGTS6B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MNP4IEIFGKLPEGVN4BLIUGTS6B/action/storage_attestation","attest_author":"https://pith.science/pith/MNP4IEIFGKLPEGVN4BLIUGTS6B/action/author_attestation","sign_citation":"https://pith.science/pith/MNP4IEIFGKLPEGVN4BLIUGTS6B/action/citation_signature","submit_replication":"https://pith.science/pith/MNP4IEIFGKLPEGVN4BLIUGTS6B/action/replication_record"}},"created_at":"2026-07-05T08:56:14.745869+00:00","updated_at":"2026-07-05T08:56:14.745869+00:00"}