{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:3UDHLNTBE62GJU2QQJINY2DJIJ","short_pith_number":"pith:3UDHLNTB","canonical_record":{"source":{"id":"2210.07636","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-10-14T08:31:45Z","cross_cats_sorted":["cs.MA"],"title_canon_sha256":"7f9fa6236d3ac33ae4883c0be7f52eef22fd32f75ac02e58c714e8fd1f8d173c","abstract_canon_sha256":"123495c62d1752b6001178fca535cad08cc1ad8ba32c059e3207e6008113bd94"},"schema_version":"1.0"},"canonical_sha256":"dd0675b66127b464d3508250dc68694265b05855eb39834020b3775ffd3937f5","source":{"kind":"arxiv","id":"2210.07636","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2210.07636","created_at":"2026-07-05T05:06:39Z"},{"alias_kind":"arxiv_version","alias_value":"2210.07636v1","created_at":"2026-07-05T05:06:39Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.07636","created_at":"2026-07-05T05:06:39Z"},{"alias_kind":"pith_short_12","alias_value":"3UDHLNTBE62G","created_at":"2026-07-05T05:06:39Z"},{"alias_kind":"pith_short_16","alias_value":"3UDHLNTBE62GJU2Q","created_at":"2026-07-05T05:06:39Z"},{"alias_kind":"pith_short_8","alias_value":"3UDHLNTB","created_at":"2026-07-05T05:06:39Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:3UDHLNTBE62GJU2QQJINY2DJIJ","target":"record","payload":{"canonical_record":{"source":{"id":"2210.07636","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-10-14T08:31:45Z","cross_cats_sorted":["cs.MA"],"title_canon_sha256":"7f9fa6236d3ac33ae4883c0be7f52eef22fd32f75ac02e58c714e8fd1f8d173c","abstract_canon_sha256":"123495c62d1752b6001178fca535cad08cc1ad8ba32c059e3207e6008113bd94"},"schema_version":"1.0"},"canonical_sha256":"dd0675b66127b464d3508250dc68694265b05855eb39834020b3775ffd3937f5","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:06:39.597671Z","signature_b64":"roGuDmpoRCl83ae43FCkTvM7jWfcME6CID5iZSv4J3WBNkZ4iI5cEwiF9Pv5ljtxzWkVflMAnIvbC5pPSqPYBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dd0675b66127b464d3508250dc68694265b05855eb39834020b3775ffd3937f5","last_reissued_at":"2026-07-05T05:06:39.597310Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:06:39.597310Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2210.07636","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:06:39Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"UnSnayYma2WTjxkFuVXKeFUva7etdD+KAzstI1JljlCKrJhVKa3Swhnr0SwT/+bou7AYV3dJNluWJkMOMkUaAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T18:35:32.153172Z"},"content_sha256":"41e512aefcd0eff622df627a5a9d75d79a455324697f4d51a0b7f5b22646b224","schema_version":"1.0","event_id":"sha256:41e512aefcd0eff622df627a5a9d75d79a455324697f4d51a0b7f5b22646b224"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:3UDHLNTBE62GJU2QQJINY2DJIJ","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Distributional Reward Estimation for Effective Multi-Agent Deep Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.LG","authors_text":"Haiyin Piao, Hechang Chen, Jifeng Hu, Lichao Sun, Sili Huang, Yanchao Sun, Yi Chang","submitted_at":"2022-10-14T08:31:45Z","abstract_excerpt":"Multi-agent reinforcement learning has drawn increasing attention in practice, e.g., robotics and automatic driving, as it can explore optimal policies using samples generated by interacting with the environment. However, high reward uncertainty still remains a problem when we want to train a satisfactory model, because obtaining high-quality reward feedback is usually expensive and even infeasible. To handle this issue, previous methods mainly focus on passive reward correction. At the same time, recent active reward estimation methods have proven to be a recipe for reducing the effect of rew"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.07636","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.07636/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:06:39Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"cQ+Hvm8cTFNQY7SISZ4BoiykBR2BAc2ZKSPdRfg7TMe047LqcpikU3qRLWKe4u9sTKfIODgrm7Fzi+xVv5BADw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T18:35:32.153666Z"},"content_sha256":"e9841ea3660dfd0820165210ce458352530a255b9108b50c114a6b7bd6ba979b","schema_version":"1.0","event_id":"sha256:e9841ea3660dfd0820165210ce458352530a255b9108b50c114a6b7bd6ba979b"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/3UDHLNTBE62GJU2QQJINY2DJIJ/bundle.json","state_url":"https://pith.science/pith/3UDHLNTBE62GJU2QQJINY2DJIJ/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/3UDHLNTBE62GJU2QQJINY2DJIJ/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T18:35:32Z","links":{"resolver":"https://pith.science/pith/3UDHLNTBE62GJU2QQJINY2DJIJ","bundle":"https://pith.science/pith/3UDHLNTBE62GJU2QQJINY2DJIJ/bundle.json","state":"https://pith.science/pith/3UDHLNTBE62GJU2QQJINY2DJIJ/state.json","well_known_bundle":"https://pith.science/.well-known/pith/3UDHLNTBE62GJU2QQJINY2DJIJ/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:3UDHLNTBE62GJU2QQJINY2DJIJ","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"123495c62d1752b6001178fca535cad08cc1ad8ba32c059e3207e6008113bd94","cross_cats_sorted":["cs.MA"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-10-14T08:31:45Z","title_canon_sha256":"7f9fa6236d3ac33ae4883c0be7f52eef22fd32f75ac02e58c714e8fd1f8d173c"},"schema_version":"1.0","source":{"id":"2210.07636","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2210.07636","created_at":"2026-07-05T05:06:39Z"},{"alias_kind":"arxiv_version","alias_value":"2210.07636v1","created_at":"2026-07-05T05:06:39Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.07636","created_at":"2026-07-05T05:06:39Z"},{"alias_kind":"pith_short_12","alias_value":"3UDHLNTBE62G","created_at":"2026-07-05T05:06:39Z"},{"alias_kind":"pith_short_16","alias_value":"3UDHLNTBE62GJU2Q","created_at":"2026-07-05T05:06:39Z"},{"alias_kind":"pith_short_8","alias_value":"3UDHLNTB","created_at":"2026-07-05T05:06:39Z"}],"graph_snapshots":[{"event_id":"sha256:e9841ea3660dfd0820165210ce458352530a255b9108b50c114a6b7bd6ba979b","target":"graph","created_at":"2026-07-05T05:06:39Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2210.07636/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Multi-agent reinforcement learning has drawn increasing attention in practice, e.g., robotics and automatic driving, as it can explore optimal policies using samples generated by interacting with the environment. However, high reward uncertainty still remains a problem when we want to train a satisfactory model, because obtaining high-quality reward feedback is usually expensive and even infeasible. To handle this issue, previous methods mainly focus on passive reward correction. At the same time, recent active reward estimation methods have proven to be a recipe for reducing the effect of rew","authors_text":"Haiyin Piao, Hechang Chen, Jifeng Hu, Lichao Sun, Sili Huang, Yanchao Sun, Yi Chang","cross_cats":["cs.MA"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-10-14T08:31:45Z","title":"Distributional Reward Estimation for Effective Multi-Agent Deep Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.07636","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:41e512aefcd0eff622df627a5a9d75d79a455324697f4d51a0b7f5b22646b224","target":"record","created_at":"2026-07-05T05:06:39Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"123495c62d1752b6001178fca535cad08cc1ad8ba32c059e3207e6008113bd94","cross_cats_sorted":["cs.MA"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-10-14T08:31:45Z","title_canon_sha256":"7f9fa6236d3ac33ae4883c0be7f52eef22fd32f75ac02e58c714e8fd1f8d173c"},"schema_version":"1.0","source":{"id":"2210.07636","kind":"arxiv","version":1}},"canonical_sha256":"dd0675b66127b464d3508250dc68694265b05855eb39834020b3775ffd3937f5","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"dd0675b66127b464d3508250dc68694265b05855eb39834020b3775ffd3937f5","first_computed_at":"2026-07-05T05:06:39.597310Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T05:06:39.597310Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"roGuDmpoRCl83ae43FCkTvM7jWfcME6CID5iZSv4J3WBNkZ4iI5cEwiF9Pv5ljtxzWkVflMAnIvbC5pPSqPYBw==","signature_status":"signed_v1","signed_at":"2026-07-05T05:06:39.597671Z","signed_message":"canonical_sha256_bytes"},"source_id":"2210.07636","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:41e512aefcd0eff622df627a5a9d75d79a455324697f4d51a0b7f5b22646b224","sha256:e9841ea3660dfd0820165210ce458352530a255b9108b50c114a6b7bd6ba979b"],"state_sha256":"8fbbd0aa13de79c25ef98588adc4f4962296f6bbf36e2f360a567438499ca776"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"mI51meJR8ZraQMmNNCiL1tbNP3NChnOk8dhdAuEwIDDnkIHt75CeLGVchRwNPFROlTRyY2sXz8mA55PyHjLBCQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T18:35:32.157077Z","bundle_sha256":"eb6abd994bc9d1f6ece749639eb33616808981121fdb2a7d07bc97a7efbc9b6b"}}