{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:B5EG6AHSLU6PFA6N3ON3GJSLAZ","short_pith_number":"pith:B5EG6AHS","schema_version":"1.0","canonical_sha256":"0f486f00f25d3cf283cddb9bb3264b0652a91617fe0d0294f2161040666c8849","source":{"kind":"arxiv","id":"2201.12122","version":3},"attestation_state":"computed","paper":{"title":"Can Wikipedia Help Offline Reinforcement Learning?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Machel Reid, Shixiang Shane Gu, Yutaro Yamada","submitted_at":"2022-01-28T13:55:35Z","abstract_excerpt":"Fine-tuning reinforcement learning (RL) models has been challenging because of a lack of large scale off-the-shelf datasets as well as high variance in transferability among different environments. Recent work has looked at tackling offline RL from the perspective of sequence modeling with improved results as result of the introduction of the Transformer architecture. However, when the model is trained from scratch, it suffers from slow convergence speeds. In this paper, we look to take advantage of this formulation of reinforcement learning as sequence modeling and investigate the transferabi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2201.12122","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-01-28T13:55:35Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"245c40c896a40bc32d79fa1f77cef45ef65524c994e60e5e62e9b3264912a32e","abstract_canon_sha256":"2ce8a2da72c13a5ba882f4a28ea4faba4dca66c7727927cc3b2eb76cbbc5d806"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:42:59.014058Z","signature_b64":"ihIZndCuxaDvMGXwYnANd8LTIWHmoFJan/1KzX4Fri9xq84eSToCio6P4rW+hFwDHD/0WVveC2ue+CsL399xCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0f486f00f25d3cf283cddb9bb3264b0652a91617fe0d0294f2161040666c8849","last_reissued_at":"2026-07-05T04:42:59.013555Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:42:59.013555Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Wikipedia Help Offline Reinforcement Learning?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Machel Reid, Shixiang Shane Gu, Yutaro Yamada","submitted_at":"2022-01-28T13:55:35Z","abstract_excerpt":"Fine-tuning reinforcement learning (RL) models has been challenging because of a lack of large scale off-the-shelf datasets as well as high variance in transferability among different environments. Recent work has looked at tackling offline RL from the perspective of sequence modeling with improved results as result of the introduction of the Transformer architecture. However, when the model is trained from scratch, it suffers from slow convergence speeds. In this paper, we look to take advantage of this formulation of reinforcement learning as sequence modeling and investigate the transferabi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2201.12122","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2201.12122/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2201.12122","created_at":"2026-07-05T04:42:59.013615+00:00"},{"alias_kind":"arxiv_version","alias_value":"2201.12122v3","created_at":"2026-07-05T04:42:59.013615+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2201.12122","created_at":"2026-07-05T04:42:59.013615+00:00"},{"alias_kind":"pith_short_12","alias_value":"B5EG6AHSLU6P","created_at":"2026-07-05T04:42:59.013615+00:00"},{"alias_kind":"pith_short_16","alias_value":"B5EG6AHSLU6PFA6N","created_at":"2026-07-05T04:42:59.013615+00:00"},{"alias_kind":"pith_short_8","alias_value":"B5EG6AHS","created_at":"2026-07-05T04:42:59.013615+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.20349","citing_title":"Naturalistic Computational Cognitive Science: Towards generalizable models and theories that capture the full range of natural behavior","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2502.20349","citing_title":"Naturalistic Computational Cognitive Science: Towards generalizable models and theories that capture the full range of natural behavior","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2205.06175","citing_title":"A Generalist Agent","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05833","citing_title":"On the Role of Language Representations in Auto-Bidding: Findings and Implications","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2204.01691","citing_title":"Do As I Can, Not As I Say: Grounding Language in Robotic Affordances","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13472","citing_title":"Bridging MARL to SARL: An Order-Independent Multi-Agent Transformer via Latent Consensus","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B5EG6AHSLU6PFA6N3ON3GJSLAZ","json":"https://pith.science/pith/B5EG6AHSLU6PFA6N3ON3GJSLAZ.json","graph_json":"https://pith.science/api/pith-number/B5EG6AHSLU6PFA6N3ON3GJSLAZ/graph.json","events_json":"https://pith.science/api/pith-number/B5EG6AHSLU6PFA6N3ON3GJSLAZ/events.json","paper":"https://pith.science/paper/B5EG6AHS"},"agent_actions":{"view_html":"https://pith.science/pith/B5EG6AHSLU6PFA6N3ON3GJSLAZ","download_json":"https://pith.science/pith/B5EG6AHSLU6PFA6N3ON3GJSLAZ.json","view_paper":"https://pith.science/paper/B5EG6AHS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2201.12122&json=true","fetch_graph":"https://pith.science/api/pith-number/B5EG6AHSLU6PFA6N3ON3GJSLAZ/graph.json","fetch_events":"https://pith.science/api/pith-number/B5EG6AHSLU6PFA6N3ON3GJSLAZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B5EG6AHSLU6PFA6N3ON3GJSLAZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B5EG6AHSLU6PFA6N3ON3GJSLAZ/action/storage_attestation","attest_author":"https://pith.science/pith/B5EG6AHSLU6PFA6N3ON3GJSLAZ/action/author_attestation","sign_citation":"https://pith.science/pith/B5EG6AHSLU6PFA6N3ON3GJSLAZ/action/citation_signature","submit_replication":"https://pith.science/pith/B5EG6AHSLU6PFA6N3ON3GJSLAZ/action/replication_record"}},"created_at":"2026-07-05T04:42:59.013615+00:00","updated_at":"2026-07-05T04:42:59.013615+00:00"}