{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:ROCWD4BOC6HAM6T3EUVZVY2DWW","short_pith_number":"pith:ROCWD4BO","schema_version":"1.0","canonical_sha256":"8b8561f02e178e067a7b252b9ae343b5a906f3e0772290e601796bb2047d4aa0","source":{"kind":"arxiv","id":"2101.05265","version":2},"attestation_state":"computed","paper":{"title":"Contrastive Behavioral Similarity Embeddings for Generalization in Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Marc G. Bellemare, Marlos C. Machado, Pablo Samuel Castro, Rishabh Agarwal","submitted_at":"2021-01-13T18:55:43Z","abstract_excerpt":"Reinforcement learning methods trained on few environments rarely learn policies that generalize to unseen environments. To improve generalization, we incorporate the inherent sequential structure in reinforcement learning into the representation learning process. This approach is orthogonal to recent approaches, which rarely exploit this structure explicitly. Specifically, we introduce a theoretically motivated policy similarity metric (PSM) for measuring behavioral similarity between states. PSM assigns high similarity to states for which the optimal policies in those states as well as in fu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2101.05265","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-01-13T18:55:43Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"b89c435b3e450e15ac22e203e8ad53649ce404e018d09ebd7288d6f89926399f","abstract_canon_sha256":"e7cf69e405dfec471d8399e8bd77019b66278fa5d1a3a1996c83dea4879bdcd6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:24:13.306872Z","signature_b64":"rGnSqEDk0JFKKJx3zI4o8OTBRcKf0iUdTmYtKWbbT/1R42KwtRbAM01xDQyN1EhQ9zOhuHt4fqkiaz7BHQNoDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8b8561f02e178e067a7b252b9ae343b5a906f3e0772290e601796bb2047d4aa0","last_reissued_at":"2026-07-05T02:24:13.306449Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:24:13.306449Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Contrastive Behavioral Similarity Embeddings for Generalization in Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Marc G. Bellemare, Marlos C. Machado, Pablo Samuel Castro, Rishabh Agarwal","submitted_at":"2021-01-13T18:55:43Z","abstract_excerpt":"Reinforcement learning methods trained on few environments rarely learn policies that generalize to unseen environments. To improve generalization, we incorporate the inherent sequential structure in reinforcement learning into the representation learning process. This approach is orthogonal to recent approaches, which rarely exploit this structure explicitly. Specifically, we introduce a theoretically motivated policy similarity metric (PSM) for measuring behavioral similarity between states. PSM assigns high similarity to states for which the optimal policies in those states as well as in fu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2101.05265","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2101.05265/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2101.05265","created_at":"2026-07-05T02:24:13.306504+00:00"},{"alias_kind":"arxiv_version","alias_value":"2101.05265v2","created_at":"2026-07-05T02:24:13.306504+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2101.05265","created_at":"2026-07-05T02:24:13.306504+00:00"},{"alias_kind":"pith_short_12","alias_value":"ROCWD4BOC6HA","created_at":"2026-07-05T02:24:13.306504+00:00"},{"alias_kind":"pith_short_16","alias_value":"ROCWD4BOC6HAM6T3","created_at":"2026-07-05T02:24:13.306504+00:00"},{"alias_kind":"pith_short_8","alias_value":"ROCWD4BO","created_at":"2026-07-05T02:24:13.306504+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.02834","citing_title":"Task-Aware Virtual Training: Enhancing Generalization in Meta-Reinforcement Learning for Out-of-Distribution Tasks","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2507.05561","citing_title":"Preemptive Solving of Future Problems: Multitask Preplay in Humans and Machines","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ROCWD4BOC6HAM6T3EUVZVY2DWW","json":"https://pith.science/pith/ROCWD4BOC6HAM6T3EUVZVY2DWW.json","graph_json":"https://pith.science/api/pith-number/ROCWD4BOC6HAM6T3EUVZVY2DWW/graph.json","events_json":"https://pith.science/api/pith-number/ROCWD4BOC6HAM6T3EUVZVY2DWW/events.json","paper":"https://pith.science/paper/ROCWD4BO"},"agent_actions":{"view_html":"https://pith.science/pith/ROCWD4BOC6HAM6T3EUVZVY2DWW","download_json":"https://pith.science/pith/ROCWD4BOC6HAM6T3EUVZVY2DWW.json","view_paper":"https://pith.science/paper/ROCWD4BO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2101.05265&json=true","fetch_graph":"https://pith.science/api/pith-number/ROCWD4BOC6HAM6T3EUVZVY2DWW/graph.json","fetch_events":"https://pith.science/api/pith-number/ROCWD4BOC6HAM6T3EUVZVY2DWW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ROCWD4BOC6HAM6T3EUVZVY2DWW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ROCWD4BOC6HAM6T3EUVZVY2DWW/action/storage_attestation","attest_author":"https://pith.science/pith/ROCWD4BOC6HAM6T3EUVZVY2DWW/action/author_attestation","sign_citation":"https://pith.science/pith/ROCWD4BOC6HAM6T3EUVZVY2DWW/action/citation_signature","submit_replication":"https://pith.science/pith/ROCWD4BOC6HAM6T3EUVZVY2DWW/action/replication_record"}},"created_at":"2026-07-05T02:24:13.306504+00:00","updated_at":"2026-07-05T02:24:13.306504+00:00"}