{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2020:B6JHMH54YOX3JO2YQZSGFIFQYG","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"9f0989b660122e126971df58c8ac00b250148b0f884229cdefea4f961dd8b44c","cross_cats_sorted":["cs.AI","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-09-29T15:48:02Z","title_canon_sha256":"17ba5af62debadcc0244785fdac51967cf3f45d47b9f1a66479613a4bed2aea9"},"schema_version":"1.0","source":{"id":"2009.14108","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2009.14108","created_at":"2026-07-05T04:35:55Z"},{"alias_kind":"arxiv_version","alias_value":"2009.14108v2","created_at":"2026-07-05T04:35:55Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2009.14108","created_at":"2026-07-05T04:35:55Z"},{"alias_kind":"pith_short_12","alias_value":"B6JHMH54YOX3","created_at":"2026-07-05T04:35:55Z"},{"alias_kind":"pith_short_16","alias_value":"B6JHMH54YOX3JO2Y","created_at":"2026-07-05T04:35:55Z"},{"alias_kind":"pith_short_8","alias_value":"B6JHMH54","created_at":"2026-07-05T04:35:55Z"}],"graph_snapshots":[{"event_id":"sha256:0fcf2000c9370d6ca2677d5f84c9dff097d5829eb9e9a0642986f5ce9bb21120","target":"graph","created_at":"2026-07-05T04:35:55Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2009.14108/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning algorithms require many samples when solving complex hierarchical tasks with sparse and delayed rewards. For such complex tasks, the recently proposed RUDDER uses reward redistribution to leverage steps in the Q-function that are associated with accomplishing sub-tasks. However, often only few episodes with high rewards are available as demonstrations since current exploration strategies cannot discover them in reasonable time. In this work, we introduce Align-RUDDER, which utilizes a profile model for reward redistribution that is obtained from multiple sequence alignme","authors_text":"Johannes Brandstetter, Jose A. Arjona-Medina, Marius-Constantin Dinu, Markus Hofmarcher, Matthias Dorfer, Patrick M. Blies, Sepp Hochreiter, Vihang P. Patil","cross_cats":["cs.AI","stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-09-29T15:48:02Z","title":"Align-RUDDER: Learning From Few Demonstrations by Reward Redistribution"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2009.14108","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:d8c4c3fe3a3f2cc75174f7f3ecc8ec2be4611ecf2f671988f21ab5c9c9ef6b77","target":"record","created_at":"2026-07-05T04:35:55Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"9f0989b660122e126971df58c8ac00b250148b0f884229cdefea4f961dd8b44c","cross_cats_sorted":["cs.AI","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-09-29T15:48:02Z","title_canon_sha256":"17ba5af62debadcc0244785fdac51967cf3f45d47b9f1a66479613a4bed2aea9"},"schema_version":"1.0","source":{"id":"2009.14108","kind":"arxiv","version":2}},"canonical_sha256":"0f92761fbcc3afb4bb58866462a0b0c18ce1842b1ba8bd1fc9ede8b51b4e1490","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"0f92761fbcc3afb4bb58866462a0b0c18ce1842b1ba8bd1fc9ede8b51b4e1490","first_computed_at":"2026-07-05T04:35:55.206731Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T04:35:55.206731Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"IEr9ZLzqcbDOW5ov1SaEcWm4pci5bnTRo25o+g9Bwwu+h0y71EnM+mrlpgAMdmEddDNdyZ44Idn9r6WktRa8DQ==","signature_status":"signed_v1","signed_at":"2026-07-05T04:35:55.207189Z","signed_message":"canonical_sha256_bytes"},"source_id":"2009.14108","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:d8c4c3fe3a3f2cc75174f7f3ecc8ec2be4611ecf2f671988f21ab5c9c9ef6b77","sha256:0fcf2000c9370d6ca2677d5f84c9dff097d5829eb9e9a0642986f5ce9bb21120"],"state_sha256":"d23a6bcc1c95cfcd396d2dde3f3332fe1b358a0b4f79c30079b964ee670261da"}