{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:IC5Z3YAMXTMATUYGLTW4DGY35T","short_pith_number":"pith:IC5Z3YAM","schema_version":"1.0","canonical_sha256":"40bb9de00cbcd809d3065cedc19b1becc0403a5d427ff9dff59761d2b5dc3d50","source":{"kind":"arxiv","id":"2208.11361","version":2},"attestation_state":"computed","paper":{"title":"Self-Supervised Exploration via Temporal Inconsistency in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Bo Ding, Dawei Feng, Huaimin Wang, Kele Xu, Xinjun Mao, Yuanzhao Zhai, Zijian Gao","submitted_at":"2022-08-24T08:19:41Z","abstract_excerpt":"Under sparse extrinsic reward settings, reinforcement learning has remained challenging, despite surging interests in this field. Previous attempts suggest that intrinsic reward can alleviate the issue caused by sparsity. In this article, we present a novel intrinsic reward that is inspired by human learning, as humans evaluate curiosity by comparing current observations with historical knowledge. Our method involves training a self-supervised prediction model, saving snapshots of the model parameters, and using nuclear norm to evaluate the temporal inconsistency between the predictions of dif"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2208.11361","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-08-24T08:19:41Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5733296581197e9fe9642a93b9920b558af9b27e77be78f1a8cb833e7d882883","abstract_canon_sha256":"f16a0ef6ba5908223348bc0f9ee0688cc3a38f8574cb8d641f618f2885942277"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:24:44.606986Z","signature_b64":"kE9HYIVN6jwRS/x47TJ+6tt9+2qbIN6doOfYOnryl/KeekzxVooWb7VpR/ecAvtSiukdeyRZ8su0dtyEQGSbAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"40bb9de00cbcd809d3065cedc19b1becc0403a5d427ff9dff59761d2b5dc3d50","last_reissued_at":"2026-07-05T06:24:44.606590Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:24:44.606590Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Self-Supervised Exploration via Temporal Inconsistency in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Bo Ding, Dawei Feng, Huaimin Wang, Kele Xu, Xinjun Mao, Yuanzhao Zhai, Zijian Gao","submitted_at":"2022-08-24T08:19:41Z","abstract_excerpt":"Under sparse extrinsic reward settings, reinforcement learning has remained challenging, despite surging interests in this field. Previous attempts suggest that intrinsic reward can alleviate the issue caused by sparsity. In this article, we present a novel intrinsic reward that is inspired by human learning, as humans evaluate curiosity by comparing current observations with historical knowledge. Our method involves training a self-supervised prediction model, saving snapshots of the model parameters, and using nuclear norm to evaluate the temporal inconsistency between the predictions of dif"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2208.11361","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2208.11361/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2208.11361","created_at":"2026-07-05T06:24:44.606651+00:00"},{"alias_kind":"arxiv_version","alias_value":"2208.11361v2","created_at":"2026-07-05T06:24:44.606651+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2208.11361","created_at":"2026-07-05T06:24:44.606651+00:00"},{"alias_kind":"pith_short_12","alias_value":"IC5Z3YAMXTMA","created_at":"2026-07-05T06:24:44.606651+00:00"},{"alias_kind":"pith_short_16","alias_value":"IC5Z3YAMXTMATUYG","created_at":"2026-07-05T06:24:44.606651+00:00"},{"alias_kind":"pith_short_8","alias_value":"IC5Z3YAM","created_at":"2026-07-05T06:24:44.606651+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IC5Z3YAMXTMATUYGLTW4DGY35T","json":"https://pith.science/pith/IC5Z3YAMXTMATUYGLTW4DGY35T.json","graph_json":"https://pith.science/api/pith-number/IC5Z3YAMXTMATUYGLTW4DGY35T/graph.json","events_json":"https://pith.science/api/pith-number/IC5Z3YAMXTMATUYGLTW4DGY35T/events.json","paper":"https://pith.science/paper/IC5Z3YAM"},"agent_actions":{"view_html":"https://pith.science/pith/IC5Z3YAMXTMATUYGLTW4DGY35T","download_json":"https://pith.science/pith/IC5Z3YAMXTMATUYGLTW4DGY35T.json","view_paper":"https://pith.science/paper/IC5Z3YAM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2208.11361&json=true","fetch_graph":"https://pith.science/api/pith-number/IC5Z3YAMXTMATUYGLTW4DGY35T/graph.json","fetch_events":"https://pith.science/api/pith-number/IC5Z3YAMXTMATUYGLTW4DGY35T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IC5Z3YAMXTMATUYGLTW4DGY35T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IC5Z3YAMXTMATUYGLTW4DGY35T/action/storage_attestation","attest_author":"https://pith.science/pith/IC5Z3YAMXTMATUYGLTW4DGY35T/action/author_attestation","sign_citation":"https://pith.science/pith/IC5Z3YAMXTMATUYGLTW4DGY35T/action/citation_signature","submit_replication":"https://pith.science/pith/IC5Z3YAMXTMATUYGLTW4DGY35T/action/replication_record"}},"created_at":"2026-07-05T06:24:44.606651+00:00","updated_at":"2026-07-05T06:24:44.606651+00:00"}