{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2021:QF6UDVO5FXOJGT5RATFINTZMTL","short_pith_number":"pith:QF6UDVO5","canonical_record":{"source":{"id":"2103.07607","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-03-13T03:26:33Z","cross_cats_sorted":[],"title_canon_sha256":"ab545d8acecf58cdb12572ee88f496e8d9d318b385fdaeb2ed708079375e8141","abstract_canon_sha256":"6bd22733a0982324a8e9986ab063c9956a9c0e7f3e30773ba80d8d2929629073"},"schema_version":"1.0"},"canonical_sha256":"817d41d5dd2ddc934fb104ca86cf2c9ad1b92a229e8a7c272f43b11a7de3e79e","source":{"kind":"arxiv","id":"2103.07607","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2103.07607","created_at":"2026-07-05T02:24:40Z"},{"alias_kind":"arxiv_version","alias_value":"2103.07607v2","created_at":"2026-07-05T02:24:40Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2103.07607","created_at":"2026-07-05T02:24:40Z"},{"alias_kind":"pith_short_12","alias_value":"QF6UDVO5FXOJ","created_at":"2026-07-05T02:24:40Z"},{"alias_kind":"pith_short_16","alias_value":"QF6UDVO5FXOJGT5R","created_at":"2026-07-05T02:24:40Z"},{"alias_kind":"pith_short_8","alias_value":"QF6UDVO5","created_at":"2026-07-05T02:24:40Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2021:QF6UDVO5FXOJGT5RATFINTZMTL","target":"record","payload":{"canonical_record":{"source":{"id":"2103.07607","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-03-13T03:26:33Z","cross_cats_sorted":[],"title_canon_sha256":"ab545d8acecf58cdb12572ee88f496e8d9d318b385fdaeb2ed708079375e8141","abstract_canon_sha256":"6bd22733a0982324a8e9986ab063c9956a9c0e7f3e30773ba80d8d2929629073"},"schema_version":"1.0"},"canonical_sha256":"817d41d5dd2ddc934fb104ca86cf2c9ad1b92a229e8a7c272f43b11a7de3e79e","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:24:40.822136Z","signature_b64":"s8prreMig7E0NGKGOGNLvmLSDxnqcvW3HQiZM4QX+2wKxJieqL5nGuAgnT4OBJ6TMf2SoHZ50uJZ6Mtm7iNaCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"817d41d5dd2ddc934fb104ca86cf2c9ad1b92a229e8a7c272f43b11a7de3e79e","last_reissued_at":"2026-07-05T02:24:40.821725Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:24:40.821725Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2103.07607","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:24:40Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"IUqygQly3Ojh20HbbKl/sn5HVdwCec/vQxDMvdJGlLLBYNNlTBBrO3+Wefk+uOjgTekdb7503Rm4FL9nIS0JDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-13T13:07:47.712175Z"},"content_sha256":"d6cb89a2e79d7d24df6003388f5d3906bea65b49bc999ac31f6911507ec4ea38","schema_version":"1.0","event_id":"sha256:d6cb89a2e79d7d24df6003388f5d3906bea65b49bc999ac31f6911507ec4ea38"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2021:QF6UDVO5FXOJGT5RATFINTZMTL","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Solving Compositional Reinforcement Learning Problems via Task Reduction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Huazhe Xu, Xiaolong Wang, Yilin Wu, Yi Wu, Yunfei Li","submitted_at":"2021-03-13T03:26:33Z","abstract_excerpt":"We propose a novel learning paradigm, Self-Imitation via Reduction (SIR), for solving compositional reinforcement learning problems. SIR is based on two core ideas: task reduction and self-imitation. Task reduction tackles a hard-to-solve task by actively reducing it to an easier task whose solution is known by the RL agent. Once the original hard task is successfully solved by task reduction, the agent naturally obtains a self-generated solution trajectory to imitate. By continuously collecting and imitating such demonstrations, the agent is able to progressively expand the solved subspace in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2103.07607","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2103.07607/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:24:40Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"5CGuBG6HZ/zBkBMYGR+VW90ZSLg/l8VjMlgGKIKOVLOAIB5GY8XhQJxXHyC87wcn6LFegMkdV62ZWgLK0L75Aw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-13T13:07:47.713232Z"},"content_sha256":"7737f5fee465bf58114ab959639fb523978ddd1e4a75bdfe08f0abf558a1bc6e","schema_version":"1.0","event_id":"sha256:7737f5fee465bf58114ab959639fb523978ddd1e4a75bdfe08f0abf558a1bc6e"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/QF6UDVO5FXOJGT5RATFINTZMTL/bundle.json","state_url":"https://pith.science/pith/QF6UDVO5FXOJGT5RATFINTZMTL/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/QF6UDVO5FXOJGT5RATFINTZMTL/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-13T13:07:47Z","links":{"resolver":"https://pith.science/pith/QF6UDVO5FXOJGT5RATFINTZMTL","bundle":"https://pith.science/pith/QF6UDVO5FXOJGT5RATFINTZMTL/bundle.json","state":"https://pith.science/pith/QF6UDVO5FXOJGT5RATFINTZMTL/state.json","well_known_bundle":"https://pith.science/.well-known/pith/QF6UDVO5FXOJGT5RATFINTZMTL/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2021:QF6UDVO5FXOJGT5RATFINTZMTL","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"6bd22733a0982324a8e9986ab063c9956a9c0e7f3e30773ba80d8d2929629073","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-03-13T03:26:33Z","title_canon_sha256":"ab545d8acecf58cdb12572ee88f496e8d9d318b385fdaeb2ed708079375e8141"},"schema_version":"1.0","source":{"id":"2103.07607","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2103.07607","created_at":"2026-07-05T02:24:40Z"},{"alias_kind":"arxiv_version","alias_value":"2103.07607v2","created_at":"2026-07-05T02:24:40Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2103.07607","created_at":"2026-07-05T02:24:40Z"},{"alias_kind":"pith_short_12","alias_value":"QF6UDVO5FXOJ","created_at":"2026-07-05T02:24:40Z"},{"alias_kind":"pith_short_16","alias_value":"QF6UDVO5FXOJGT5R","created_at":"2026-07-05T02:24:40Z"},{"alias_kind":"pith_short_8","alias_value":"QF6UDVO5","created_at":"2026-07-05T02:24:40Z"}],"graph_snapshots":[{"event_id":"sha256:7737f5fee465bf58114ab959639fb523978ddd1e4a75bdfe08f0abf558a1bc6e","target":"graph","created_at":"2026-07-05T02:24:40Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2103.07607/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We propose a novel learning paradigm, Self-Imitation via Reduction (SIR), for solving compositional reinforcement learning problems. SIR is based on two core ideas: task reduction and self-imitation. Task reduction tackles a hard-to-solve task by actively reducing it to an easier task whose solution is known by the RL agent. Once the original hard task is successfully solved by task reduction, the agent naturally obtains a self-generated solution trajectory to imitate. By continuously collecting and imitating such demonstrations, the agent is able to progressively expand the solved subspace in","authors_text":"Huazhe Xu, Xiaolong Wang, Yilin Wu, Yi Wu, Yunfei Li","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-03-13T03:26:33Z","title":"Solving Compositional Reinforcement Learning Problems via Task Reduction"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2103.07607","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:d6cb89a2e79d7d24df6003388f5d3906bea65b49bc999ac31f6911507ec4ea38","target":"record","created_at":"2026-07-05T02:24:40Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"6bd22733a0982324a8e9986ab063c9956a9c0e7f3e30773ba80d8d2929629073","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-03-13T03:26:33Z","title_canon_sha256":"ab545d8acecf58cdb12572ee88f496e8d9d318b385fdaeb2ed708079375e8141"},"schema_version":"1.0","source":{"id":"2103.07607","kind":"arxiv","version":2}},"canonical_sha256":"817d41d5dd2ddc934fb104ca86cf2c9ad1b92a229e8a7c272f43b11a7de3e79e","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"817d41d5dd2ddc934fb104ca86cf2c9ad1b92a229e8a7c272f43b11a7de3e79e","first_computed_at":"2026-07-05T02:24:40.821725Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T02:24:40.821725Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"s8prreMig7E0NGKGOGNLvmLSDxnqcvW3HQiZM4QX+2wKxJieqL5nGuAgnT4OBJ6TMf2SoHZ50uJZ6Mtm7iNaCg==","signature_status":"signed_v1","signed_at":"2026-07-05T02:24:40.822136Z","signed_message":"canonical_sha256_bytes"},"source_id":"2103.07607","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:d6cb89a2e79d7d24df6003388f5d3906bea65b49bc999ac31f6911507ec4ea38","sha256:7737f5fee465bf58114ab959639fb523978ddd1e4a75bdfe08f0abf558a1bc6e"],"state_sha256":"d1e4c3d886a9cd1cb566dc0ee7714c4ab8b70ad7e8b29f3d09b4a14da7e23832"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"/D99bze53U9oVy8XO3TDwGlrgw0B+Q/yY+MLKyl+543nX5MTNzNdTKL0vkr/J/ysYb2dC/FnWq404C6Esn2QDw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-13T13:07:47.720312Z","bundle_sha256":"e0a68c598f9370ebfabe14746afb530e130f2163e2229057eb63638d37fcf3a1"}}