{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:WZYAEPMVVBR7FFCBNBGQ2N5AF2","short_pith_number":"pith:WZYAEPMV","canonical_record":{"source":{"id":"2203.03535","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-03-07T17:32:35Z","cross_cats_sorted":["cs.AI","cs.MA"],"title_canon_sha256":"e3d2265cec473362b028f65fd53410aed94064f55cb415be0b897571176829d5","abstract_canon_sha256":"f0f1af45ac9f229c453fb28012d9819e5024c30bce4bfd5d51f5116bbd73f05a"},"schema_version":"1.0"},"canonical_sha256":"b670023d95a863f29441684d0d37a02e96529c99a02328bbdcb2dc79ba59c887","source":{"kind":"arxiv","id":"2203.03535","version":4},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2203.03535","created_at":"2026-07-05T05:06:55Z"},{"alias_kind":"arxiv_version","alias_value":"2203.03535v4","created_at":"2026-07-05T05:06:55Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.03535","created_at":"2026-07-05T05:06:55Z"},{"alias_kind":"pith_short_12","alias_value":"WZYAEPMVVBR7","created_at":"2026-07-05T05:06:55Z"},{"alias_kind":"pith_short_16","alias_value":"WZYAEPMVVBR7FFCB","created_at":"2026-07-05T05:06:55Z"},{"alias_kind":"pith_short_8","alias_value":"WZYAEPMV","created_at":"2026-07-05T05:06:55Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:WZYAEPMVVBR7FFCBNBGQ2N5AF2","target":"record","payload":{"canonical_record":{"source":{"id":"2203.03535","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-03-07T17:32:35Z","cross_cats_sorted":["cs.AI","cs.MA"],"title_canon_sha256":"e3d2265cec473362b028f65fd53410aed94064f55cb415be0b897571176829d5","abstract_canon_sha256":"f0f1af45ac9f229c453fb28012d9819e5024c30bce4bfd5d51f5116bbd73f05a"},"schema_version":"1.0"},"canonical_sha256":"b670023d95a863f29441684d0d37a02e96529c99a02328bbdcb2dc79ba59c887","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:06:55.591709Z","signature_b64":"Q4uXbJIuRGfdOVkNg0wuLqNOwY7Rce6y8eR+gx0gbFtOiTkC7Skeriouze08cdwkPL1E3bxjMbbF/eHFbLW5Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b670023d95a863f29441684d0d37a02e96529c99a02328bbdcb2dc79ba59c887","last_reissued_at":"2026-07-05T05:06:55.591252Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:06:55.591252Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2203.03535","source_version":4,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:06:55Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"pAmUMlySNLELiNFQeNc5IQXzSLaXURMmimsxR4wWY69lcg87TiypAdlEaRbNQqpciiOamWftMIMQA5ObiOz0Aw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-05T14:30:59.197211Z"},"content_sha256":"c9dd01ecf26780b9c6f17fc929888093de5a5085317350bdc283749b5dfb15a8","schema_version":"1.0","event_id":"sha256:c9dd01ecf26780b9c6f17fc929888093de5a5085317350bdc283749b5dfb15a8"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:WZYAEPMVVBR7FFCBNBGQ2N5AF2","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Influencing Long-Term Behavior in Multiagent Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.MA"],"primary_cat":"cs.LG","authors_text":"Chuangchuang Sun, Dong-Ki Kim, Gerald Tesauro, Jakob N. Foerster, Jonathan P. How, Matthew Riemer, Miao Liu, Michael Everett","submitted_at":"2022-03-07T17:32:35Z","abstract_excerpt":"The main challenge of multiagent reinforcement learning is the difficulty of learning useful policies in the presence of other simultaneously learning agents whose changing behaviors jointly affect the environment's transition and reward dynamics. An effective approach that has recently emerged for addressing this non-stationarity is for each agent to anticipate the learning of other agents and influence the evolution of future policies towards desirable behavior for its own benefit. Unfortunately, previous approaches for achieving this suffer from myopic evaluation, considering only a finite "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.03535","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.03535/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:06:55Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"0gBn0EAlfDpzK8yHlNTbOPr+hJYcZhLZC64wfBLmA5j9GZ6f+LJXKLAmJCjMfrKWSrJ/h8MpFn5FYl99jQssAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-05T14:30:59.198185Z"},"content_sha256":"523873a4e192bc7bfe3187e6223339731a578f3433452cbf3751a7b0e12bfb26","schema_version":"1.0","event_id":"sha256:523873a4e192bc7bfe3187e6223339731a578f3433452cbf3751a7b0e12bfb26"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/WZYAEPMVVBR7FFCBNBGQ2N5AF2/bundle.json","state_url":"https://pith.science/pith/WZYAEPMVVBR7FFCBNBGQ2N5AF2/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/WZYAEPMVVBR7FFCBNBGQ2N5AF2/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-07-05T14:30:59Z","links":{"resolver":"https://pith.science/pith/WZYAEPMVVBR7FFCBNBGQ2N5AF2","bundle":"https://pith.science/pith/WZYAEPMVVBR7FFCBNBGQ2N5AF2/bundle.json","state":"https://pith.science/pith/WZYAEPMVVBR7FFCBNBGQ2N5AF2/state.json","well_known_bundle":"https://pith.science/.well-known/pith/WZYAEPMVVBR7FFCBNBGQ2N5AF2/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:WZYAEPMVVBR7FFCBNBGQ2N5AF2","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"f0f1af45ac9f229c453fb28012d9819e5024c30bce4bfd5d51f5116bbd73f05a","cross_cats_sorted":["cs.AI","cs.MA"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-03-07T17:32:35Z","title_canon_sha256":"e3d2265cec473362b028f65fd53410aed94064f55cb415be0b897571176829d5"},"schema_version":"1.0","source":{"id":"2203.03535","kind":"arxiv","version":4}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2203.03535","created_at":"2026-07-05T05:06:55Z"},{"alias_kind":"arxiv_version","alias_value":"2203.03535v4","created_at":"2026-07-05T05:06:55Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.03535","created_at":"2026-07-05T05:06:55Z"},{"alias_kind":"pith_short_12","alias_value":"WZYAEPMVVBR7","created_at":"2026-07-05T05:06:55Z"},{"alias_kind":"pith_short_16","alias_value":"WZYAEPMVVBR7FFCB","created_at":"2026-07-05T05:06:55Z"},{"alias_kind":"pith_short_8","alias_value":"WZYAEPMV","created_at":"2026-07-05T05:06:55Z"}],"graph_snapshots":[{"event_id":"sha256:523873a4e192bc7bfe3187e6223339731a578f3433452cbf3751a7b0e12bfb26","target":"graph","created_at":"2026-07-05T05:06:55Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2203.03535/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The main challenge of multiagent reinforcement learning is the difficulty of learning useful policies in the presence of other simultaneously learning agents whose changing behaviors jointly affect the environment's transition and reward dynamics. An effective approach that has recently emerged for addressing this non-stationarity is for each agent to anticipate the learning of other agents and influence the evolution of future policies towards desirable behavior for its own benefit. Unfortunately, previous approaches for achieving this suffer from myopic evaluation, considering only a finite ","authors_text":"Chuangchuang Sun, Dong-Ki Kim, Gerald Tesauro, Jakob N. Foerster, Jonathan P. How, Matthew Riemer, Miao Liu, Michael Everett","cross_cats":["cs.AI","cs.MA"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-03-07T17:32:35Z","title":"Influencing Long-Term Behavior in Multiagent Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.03535","kind":"arxiv","version":4},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:c9dd01ecf26780b9c6f17fc929888093de5a5085317350bdc283749b5dfb15a8","target":"record","created_at":"2026-07-05T05:06:55Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"f0f1af45ac9f229c453fb28012d9819e5024c30bce4bfd5d51f5116bbd73f05a","cross_cats_sorted":["cs.AI","cs.MA"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-03-07T17:32:35Z","title_canon_sha256":"e3d2265cec473362b028f65fd53410aed94064f55cb415be0b897571176829d5"},"schema_version":"1.0","source":{"id":"2203.03535","kind":"arxiv","version":4}},"canonical_sha256":"b670023d95a863f29441684d0d37a02e96529c99a02328bbdcb2dc79ba59c887","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b670023d95a863f29441684d0d37a02e96529c99a02328bbdcb2dc79ba59c887","first_computed_at":"2026-07-05T05:06:55.591252Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T05:06:55.591252Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"Q4uXbJIuRGfdOVkNg0wuLqNOwY7Rce6y8eR+gx0gbFtOiTkC7Skeriouze08cdwkPL1E3bxjMbbF/eHFbLW5Dw==","signature_status":"signed_v1","signed_at":"2026-07-05T05:06:55.591709Z","signed_message":"canonical_sha256_bytes"},"source_id":"2203.03535","source_kind":"arxiv","source_version":4}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:c9dd01ecf26780b9c6f17fc929888093de5a5085317350bdc283749b5dfb15a8","sha256:523873a4e192bc7bfe3187e6223339731a578f3433452cbf3751a7b0e12bfb26"],"state_sha256":"901e5b2bc7023e0da908d404a9ebf0046ef9ca0c6a75b94b982c3da507c6b343"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Lzvb0bIdLaFlpGs7XZilPE9+kaHzdBmthVLiflerYGCiFcFNAoqrg8BQwt9qydfDfGT3ihProQShJuTohlqqCQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-07-05T14:30:59.202533Z","bundle_sha256":"8a9d2b37e7e9a1162a4b98829419ef28fb3f77e535f75cc5c570961b72a63d9a"}}