{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JXDFSOFH3MXDSN62KAR5JHXIRR","short_pith_number":"pith:JXDFSOFH","schema_version":"1.0","canonical_sha256":"4dc65938a7db2e3937da5023d49ee88c668b2c15734867b2e7b2463e2bab50f5","source":{"kind":"arxiv","id":"2405.12001","version":4},"attestation_state":"computed","paper":{"title":"Scrutinize What We Ignore: Reining In Task Representation Shift Of Context-Based Offline Meta Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Anqi Guo, Boyuan Zheng, Hai Zhang, Jinhang Liu, Junqiao Zhao, Lanqing Li, Tianying Ji","submitted_at":"2024-05-20T13:14:26Z","abstract_excerpt":"Offline meta reinforcement learning (OMRL) has emerged as a promising approach for interaction avoidance and strong generalization performance by leveraging pre-collected data and meta-learning techniques. Previous context-based approaches predominantly rely on the intuition that alternating optimization between the context encoder and the policy can lead to performance improvements, as long as the context encoder follows the principle of maximizing the mutual information between the task variable $M$ and its latent representation $Z$ ($I(Z;M)$) while the policy adopts the standard offline rei"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.12001","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-20T13:14:26Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4801d2081c674226818d84557ed9cbb442e3dbb1af57bcc1bbaf6ebc67b8971f","abstract_canon_sha256":"49f9706669f026d5858bbd3ea30cfc86419a4098bc2486f6344944781ca17e36"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:08:16.778424Z","signature_b64":"bw3D1GCtPOCNth9gnckZFZPRy29r+5mfoF18rtc2n1dkDYlC4KRWXQi18vnfAi7BA72HVq3wNmLxd1RnGuQQCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4dc65938a7db2e3937da5023d49ee88c668b2c15734867b2e7b2463e2bab50f5","last_reissued_at":"2026-07-05T10:08:16.777998Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:08:16.777998Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scrutinize What We Ignore: Reining In Task Representation Shift Of Context-Based Offline Meta Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Anqi Guo, Boyuan Zheng, Hai Zhang, Jinhang Liu, Junqiao Zhao, Lanqing Li, Tianying Ji","submitted_at":"2024-05-20T13:14:26Z","abstract_excerpt":"Offline meta reinforcement learning (OMRL) has emerged as a promising approach for interaction avoidance and strong generalization performance by leveraging pre-collected data and meta-learning techniques. Previous context-based approaches predominantly rely on the intuition that alternating optimization between the context encoder and the policy can lead to performance improvements, as long as the context encoder follows the principle of maximizing the mutual information between the task variable $M$ and its latent representation $Z$ ($I(Z;M)$) while the policy adopts the standard offline rei"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.12001","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.12001/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.12001","created_at":"2026-07-05T10:08:16.778055+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.12001v4","created_at":"2026-07-05T10:08:16.778055+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.12001","created_at":"2026-07-05T10:08:16.778055+00:00"},{"alias_kind":"pith_short_12","alias_value":"JXDFSOFH3MXD","created_at":"2026-07-05T10:08:16.778055+00:00"},{"alias_kind":"pith_short_16","alias_value":"JXDFSOFH3MXDSN62","created_at":"2026-07-05T10:08:16.778055+00:00"},{"alias_kind":"pith_short_8","alias_value":"JXDFSOFH","created_at":"2026-07-05T10:08:16.778055+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JXDFSOFH3MXDSN62KAR5JHXIRR","json":"https://pith.science/pith/JXDFSOFH3MXDSN62KAR5JHXIRR.json","graph_json":"https://pith.science/api/pith-number/JXDFSOFH3MXDSN62KAR5JHXIRR/graph.json","events_json":"https://pith.science/api/pith-number/JXDFSOFH3MXDSN62KAR5JHXIRR/events.json","paper":"https://pith.science/paper/JXDFSOFH"},"agent_actions":{"view_html":"https://pith.science/pith/JXDFSOFH3MXDSN62KAR5JHXIRR","download_json":"https://pith.science/pith/JXDFSOFH3MXDSN62KAR5JHXIRR.json","view_paper":"https://pith.science/paper/JXDFSOFH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.12001&json=true","fetch_graph":"https://pith.science/api/pith-number/JXDFSOFH3MXDSN62KAR5JHXIRR/graph.json","fetch_events":"https://pith.science/api/pith-number/JXDFSOFH3MXDSN62KAR5JHXIRR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JXDFSOFH3MXDSN62KAR5JHXIRR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JXDFSOFH3MXDSN62KAR5JHXIRR/action/storage_attestation","attest_author":"https://pith.science/pith/JXDFSOFH3MXDSN62KAR5JHXIRR/action/author_attestation","sign_citation":"https://pith.science/pith/JXDFSOFH3MXDSN62KAR5JHXIRR/action/citation_signature","submit_replication":"https://pith.science/pith/JXDFSOFH3MXDSN62KAR5JHXIRR/action/replication_record"}},"created_at":"2026-07-05T10:08:16.778055+00:00","updated_at":"2026-07-05T10:08:16.778055+00:00"}