{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:SDW2CLQTDAFVYWEYINR5DTZGMO","short_pith_number":"pith:SDW2CLQT","canonical_record":{"source":{"id":"2505.20579","kind":"arxiv","version":7},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-26T23:28:52Z","cross_cats_sorted":["cs.AI","cs.MA"],"title_canon_sha256":"607afafa99b172690fe27c68b4f874a939b6b7740991472d9ffc542eb52e7199","abstract_canon_sha256":"229c863fbdf52527e57a767b1f868dbf5d00f7e980e47c981e4182dda0823a79"},"schema_version":"1.0"},"canonical_sha256":"90eda12e13180b5c58984363d1cf2663a7110f601102668e0c59e79efc69b71c","source":{"kind":"arxiv","id":"2505.20579","version":7},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.20579","created_at":"2026-07-21T01:20:33Z"},{"alias_kind":"arxiv_version","alias_value":"2505.20579v7","created_at":"2026-07-21T01:20:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.20579","created_at":"2026-07-21T01:20:33Z"},{"alias_kind":"pith_short_12","alias_value":"SDW2CLQTDAFV","created_at":"2026-07-21T01:20:33Z"},{"alias_kind":"pith_short_16","alias_value":"SDW2CLQTDAFVYWEY","created_at":"2026-07-21T01:20:33Z"},{"alias_kind":"pith_short_8","alias_value":"SDW2CLQT","created_at":"2026-07-21T01:20:33Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:SDW2CLQTDAFVYWEYINR5DTZGMO","target":"record","payload":{"canonical_record":{"source":{"id":"2505.20579","kind":"arxiv","version":7},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-26T23:28:52Z","cross_cats_sorted":["cs.AI","cs.MA"],"title_canon_sha256":"607afafa99b172690fe27c68b4f874a939b6b7740991472d9ffc542eb52e7199","abstract_canon_sha256":"229c863fbdf52527e57a767b1f868dbf5d00f7e980e47c981e4182dda0823a79"},"schema_version":"1.0"},"canonical_sha256":"90eda12e13180b5c58984363d1cf2663a7110f601102668e0c59e79efc69b71c","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-21T01:20:33.198221Z","signature_b64":"KlDhd1FenGbz3kH3LgrGzlsMTIp+U/qZG4+eufDA9bSw+3W8I6FSqhdZ2kZxxEYkoKk4zjhqgQJhvZs/h8TwDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"90eda12e13180b5c58984363d1cf2663a7110f601102668e0c59e79efc69b71c","last_reissued_at":"2026-07-21T01:20:33.197163Z","signature_status":"signed_v1","first_computed_at":"2026-07-21T01:20:33.197163Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2505.20579","source_version":7,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-21T01:20:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"eJHi6GUCJmwEX7vd7YfH/La6yotwWIGy2QSSk5SFXehgBvMvmgVv9gb/9vArKRUkcxtYFIpTiKHjTqZnSNCJDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-13T23:26:10.004926Z"},"content_sha256":"829c8fa35ac5c641a75724788a3de628dac2b8958a3834ac64cb3d70e14643c7","schema_version":"1.0","event_id":"sha256:829c8fa35ac5c641a75724788a3de628dac2b8958a3834ac64cb3d70e14643c7"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:SDW2CLQTDAFVYWEYINR5DTZGMO","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"The challenge of hidden gifts in multi-agent reinforcement learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"Many state-of-the-art multi-agent reinforcement learning algorithms fail to learn collective rewards when success depends on hidden gifts from other agents' unobserved actions.","cross_cats":["cs.AI","cs.MA"],"primary_cat":"cs.LG","authors_text":"Blake A. Richards, Dane Malenfant","submitted_at":"2025-05-26T23:28:52Z","abstract_excerpt":"Sometimes we benefit from actions that others have taken even when we are unaware that they took those actions. For example, if your neighbor chooses not to take a parking spot in front of your house when you are not there, you can benefit, even without being aware that they took this action. These ``hidden gifts'' represent an interesting challenge for multi-agent reinforcement learning (MARL), since assigning credit when the beneficial actions of others are hidden is non-trivial. Here, we study the impact of hidden gifts with a simple MARL task. In this task, agents in a grid-world environme"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Several different state-of-the-art MARL algorithms, including MARL specific architectures, fail to learn how to obtain the collective reward in this simple task; decentralized actor-critic policy gradient agents can succeed when provided with information about their own action history, and a derived correction term for policy gradient agents reduces the variance in learning and helps them to converge to collective success more reliably.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the complete absence of any signal indicating other agents have dropped the key is the decisive factor causing MARL algorithms to fail at collective rewards, rather than other details of the grid-world dynamics, reward scaling, or training hyperparameters.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"In a grid-world task with one shared key and no feedback on others' drops, SOTA MARL algorithms fail at collective rewards but decentralized policy gradients with action history and a derived correction term succeed.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Many state-of-the-art multi-agent reinforcement learning algorithms fail to learn collective rewards when success depends on hidden gifts from other agents' unobserved actions.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"dfd550b968cfca3f889974831ba24db4e8318a79ca9fc54052bbf3a3fcd3a600"},"source":{"id":"2505.20579","kind":"arxiv","version":7},"verdict":{"id":"08940957-bb20-41c6-a0d1-42baf07edb6c","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-19T12:20:08.889367Z","strongest_claim":"Several different state-of-the-art MARL algorithms, including MARL specific architectures, fail to learn how to obtain the collective reward in this simple task; decentralized actor-critic policy gradient agents can succeed when provided with information about their own action history, and a derived correction term for policy gradient agents reduces the variance in learning and helps them to converge to collective success more reliably.","one_line_summary":"In a grid-world task with one shared key and no feedback on others' drops, SOTA MARL algorithms fail at collective rewards but decentralized policy gradients with action history and a derived correction term succeed.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the complete absence of any signal indicating other agents have dropped the key is the decisive factor causing MARL algorithms to fail at collective rewards, rather than other details of the grid-world dynamics, reward scaling, or training hyperparameters.","pith_extraction_headline":"Many state-of-the-art multi-agent reinforcement learning algorithms fail to learn collective rewards when success depends on hidden gifts from other agents' unobserved actions."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.20579/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":2,"snapshot_sha256":"b536092e6622ec2270155cbfff28d69324bfd48f75128a5aeb58132d8f4bef13"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"08940957-bb20-41c6-a0d1-42baf07edb6c"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-21T01:20:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"IQSiQBRWAUZNjZjjHnQYUm00uNu4huJ09rBM+feQH9GfjJaxK/K2kc7jSFsAtx/XZj0kFvDpsLvzLXoT+DExBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-13T23:26:10.006543Z"},"content_sha256":"5250189040335f62cb8760fea8cb5a64a4b67955a6c52b00ba232161aa82b6a9","schema_version":"1.0","event_id":"sha256:5250189040335f62cb8760fea8cb5a64a4b67955a6c52b00ba232161aa82b6a9"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/SDW2CLQTDAFVYWEYINR5DTZGMO/bundle.json","state_url":"https://pith.science/pith/SDW2CLQTDAFVYWEYINR5DTZGMO/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/SDW2CLQTDAFVYWEYINR5DTZGMO/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-13T23:26:10Z","links":{"resolver":"https://pith.science/pith/SDW2CLQTDAFVYWEYINR5DTZGMO","bundle":"https://pith.science/pith/SDW2CLQTDAFVYWEYINR5DTZGMO/bundle.json","state":"https://pith.science/pith/SDW2CLQTDAFVYWEYINR5DTZGMO/state.json","well_known_bundle":"https://pith.science/.well-known/pith/SDW2CLQTDAFVYWEYINR5DTZGMO/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:SDW2CLQTDAFVYWEYINR5DTZGMO","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"229c863fbdf52527e57a767b1f868dbf5d00f7e980e47c981e4182dda0823a79","cross_cats_sorted":["cs.AI","cs.MA"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-26T23:28:52Z","title_canon_sha256":"607afafa99b172690fe27c68b4f874a939b6b7740991472d9ffc542eb52e7199"},"schema_version":"1.0","source":{"id":"2505.20579","kind":"arxiv","version":7}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.20579","created_at":"2026-07-21T01:20:33Z"},{"alias_kind":"arxiv_version","alias_value":"2505.20579v7","created_at":"2026-07-21T01:20:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.20579","created_at":"2026-07-21T01:20:33Z"},{"alias_kind":"pith_short_12","alias_value":"SDW2CLQTDAFV","created_at":"2026-07-21T01:20:33Z"},{"alias_kind":"pith_short_16","alias_value":"SDW2CLQTDAFVYWEY","created_at":"2026-07-21T01:20:33Z"},{"alias_kind":"pith_short_8","alias_value":"SDW2CLQT","created_at":"2026-07-21T01:20:33Z"}],"graph_snapshots":[{"event_id":"sha256:5250189040335f62cb8760fea8cb5a64a4b67955a6c52b00ba232161aa82b6a9","target":"graph","created_at":"2026-07-21T01:20:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Several different state-of-the-art MARL algorithms, including MARL specific architectures, fail to learn how to obtain the collective reward in this simple task; decentralized actor-critic policy gradient agents can succeed when provided with information about their own action history, and a derived correction term for policy gradient agents reduces the variance in learning and helps them to converge to collective success more reliably."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That the complete absence of any signal indicating other agents have dropped the key is the decisive factor causing MARL algorithms to fail at collective rewards, rather than other details of the grid-world dynamics, reward scaling, or training hyperparameters."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"In a grid-world task with one shared key and no feedback on others' drops, SOTA MARL algorithms fail at collective rewards but decentralized policy gradients with action history and a derived correction term succeed."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Many state-of-the-art multi-agent reinforcement learning algorithms fail to learn collective rewards when success depends on hidden gifts from other agents' unobserved actions."}],"snapshot_sha256":"dfd550b968cfca3f889974831ba24db4e8318a79ca9fc54052bbf3a3fcd3a600"},"formal_canon":{"evidence_count":2,"snapshot_sha256":"b536092e6622ec2270155cbfff28d69324bfd48f75128a5aeb58132d8f4bef13"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2505.20579/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Sometimes we benefit from actions that others have taken even when we are unaware that they took those actions. For example, if your neighbor chooses not to take a parking spot in front of your house when you are not there, you can benefit, even without being aware that they took this action. These ``hidden gifts'' represent an interesting challenge for multi-agent reinforcement learning (MARL), since assigning credit when the beneficial actions of others are hidden is non-trivial. Here, we study the impact of hidden gifts with a simple MARL task. In this task, agents in a grid-world environme","authors_text":"Blake A. Richards, Dane Malenfant","cross_cats":["cs.AI","cs.MA"],"headline":"Many state-of-the-art multi-agent reinforcement learning algorithms fail to learn collective rewards when success depends on hidden gifts from other agents' unobserved actions.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-26T23:28:52Z","title":"The challenge of hidden gifts in multi-agent reinforcement learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.20579","kind":"arxiv","version":7},"verdict":{"created_at":"2026-05-19T12:20:08.889367Z","id":"08940957-bb20-41c6-a0d1-42baf07edb6c","model_set":{"reader":"grok-4.3"},"one_line_summary":"In a grid-world task with one shared key and no feedback on others' drops, SOTA MARL algorithms fail at collective rewards but decentralized policy gradients with action history and a derived correction term succeed.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Many state-of-the-art multi-agent reinforcement learning algorithms fail to learn collective rewards when success depends on hidden gifts from other agents' unobserved actions.","strongest_claim":"Several different state-of-the-art MARL algorithms, including MARL specific architectures, fail to learn how to obtain the collective reward in this simple task; decentralized actor-critic policy gradient agents can succeed when provided with information about their own action history, and a derived correction term for policy gradient agents reduces the variance in learning and helps them to converge to collective success more reliably.","weakest_assumption":"That the complete absence of any signal indicating other agents have dropped the key is the decisive factor causing MARL algorithms to fail at collective rewards, rather than other details of the grid-world dynamics, reward scaling, or training hyperparameters."}},"verdict_id":"08940957-bb20-41c6-a0d1-42baf07edb6c"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:829c8fa35ac5c641a75724788a3de628dac2b8958a3834ac64cb3d70e14643c7","target":"record","created_at":"2026-07-21T01:20:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"229c863fbdf52527e57a767b1f868dbf5d00f7e980e47c981e4182dda0823a79","cross_cats_sorted":["cs.AI","cs.MA"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-26T23:28:52Z","title_canon_sha256":"607afafa99b172690fe27c68b4f874a939b6b7740991472d9ffc542eb52e7199"},"schema_version":"1.0","source":{"id":"2505.20579","kind":"arxiv","version":7}},"canonical_sha256":"90eda12e13180b5c58984363d1cf2663a7110f601102668e0c59e79efc69b71c","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"90eda12e13180b5c58984363d1cf2663a7110f601102668e0c59e79efc69b71c","first_computed_at":"2026-07-21T01:20:33.197163Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-21T01:20:33.197163Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"KlDhd1FenGbz3kH3LgrGzlsMTIp+U/qZG4+eufDA9bSw+3W8I6FSqhdZ2kZxxEYkoKk4zjhqgQJhvZs/h8TwDQ==","signature_status":"signed_v1","signed_at":"2026-07-21T01:20:33.198221Z","signed_message":"canonical_sha256_bytes"},"source_id":"2505.20579","source_kind":"arxiv","source_version":7}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:829c8fa35ac5c641a75724788a3de628dac2b8958a3834ac64cb3d70e14643c7","sha256:5250189040335f62cb8760fea8cb5a64a4b67955a6c52b00ba232161aa82b6a9"],"state_sha256":"9b571bf2a9e22a3f082ed493e5ae5a2ddfd1f78593f93c91440d13370e25848b"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"SKCgAzALjm1mwgC2JvAPdIFu/ukuQFlg/YQ1A3W3QGWNbHdazfq5cN5qxWjR9Bx3e7XyWQzcOndC0MRpIVtcCg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-13T23:26:10.015668Z","bundle_sha256":"018904a3c82b458d490398c5b9e87ab9fad6c3f09f6012153788722ca329bb38"}}