{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:BO6O7YM6H75O5KSFKRVGWHJPXO","short_pith_number":"pith:BO6O7YM6","schema_version":"1.0","canonical_sha256":"0bbcefe19e3ffaeeaa45546a6b1d2fbba6a542ba9dc32d39b837af5b755f6714","source":{"kind":"arxiv","id":"1809.05096","version":3},"attestation_state":"computed","paper":{"title":"Negative Update Intervals in Deep Multi-Agent Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.MA","authors_text":"Gregory Palmer, Karl Tuyls, Rahul Savani","submitted_at":"2018-09-13T15:46:55Z","abstract_excerpt":"In Multi-Agent Reinforcement Learning (MA-RL), independent cooperative learners must overcome a number of pathologies to learn optimal joint policies. Addressing one pathology often leaves approaches vulnerable towards others. For instance, hysteretic Q-learning addresses miscoordination while leaving agents vulnerable towards misleading stochastic rewards. Other methods, such as leniency, have proven more robust when dealing with multiple pathologies simultaneously. However, leniency has predominately been studied within the context of strategic form games (bimatrix games) and fully observabl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1809.05096","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.MA","submitted_at":"2018-09-13T15:46:55Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"b1e1e71f9d65d0ed92efe2f29dd0a7f08e49fce6753dc7c2dedfd4b364aeb9ef","abstract_canon_sha256":"deea828fe6f2067355e67b5d830f14a2d7514e419cfc5b8f090cb620176f4946"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-17T23:46:55.152414Z","signature_b64":"ei3oRhjlpzvIAiNz7BhKa0d9UhukX66PY5LjoSVkKYqonUuq/f9XAydBiQKLlPF2KbCZKLAlllmtxwdQm+O/BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0bbcefe19e3ffaeeaa45546a6b1d2fbba6a542ba9dc32d39b837af5b755f6714","last_reissued_at":"2026-05-17T23:46:55.151756Z","signature_status":"signed_v1","first_computed_at":"2026-05-17T23:46:55.151756Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Negative Update Intervals in Deep Multi-Agent Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.MA","authors_text":"Gregory Palmer, Karl Tuyls, Rahul Savani","submitted_at":"2018-09-13T15:46:55Z","abstract_excerpt":"In Multi-Agent Reinforcement Learning (MA-RL), independent cooperative learners must overcome a number of pathologies to learn optimal joint policies. Addressing one pathology often leaves approaches vulnerable towards others. For instance, hysteretic Q-learning addresses miscoordination while leaving agents vulnerable towards misleading stochastic rewards. Other methods, such as leniency, have proven more robust when dealing with multiple pathologies simultaneously. However, leniency has predominately been studied within the context of strategic form games (bimatrix games) and fully observabl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1809.05096","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1809.05096","created_at":"2026-05-17T23:46:55.151835+00:00"},{"alias_kind":"arxiv_version","alias_value":"1809.05096v3","created_at":"2026-05-17T23:46:55.151835+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1809.05096","created_at":"2026-05-17T23:46:55.151835+00:00"},{"alias_kind":"pith_short_12","alias_value":"BO6O7YM6H75O","created_at":"2026-05-18T12:32:16.446611+00:00"},{"alias_kind":"pith_short_16","alias_value":"BO6O7YM6H75O5KSF","created_at":"2026-05-18T12:32:16.446611+00:00"},{"alias_kind":"pith_short_8","alias_value":"BO6O7YM6","created_at":"2026-05-18T12:32:16.446611+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2411.11099","citing_title":"Mitigating Relative Over-Generalization in Multi-Agent Reinforcement Learning","ref_index":21,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BO6O7YM6H75O5KSFKRVGWHJPXO","json":"https://pith.science/pith/BO6O7YM6H75O5KSFKRVGWHJPXO.json","graph_json":"https://pith.science/api/pith-number/BO6O7YM6H75O5KSFKRVGWHJPXO/graph.json","events_json":"https://pith.science/api/pith-number/BO6O7YM6H75O5KSFKRVGWHJPXO/events.json","paper":"https://pith.science/paper/BO6O7YM6"},"agent_actions":{"view_html":"https://pith.science/pith/BO6O7YM6H75O5KSFKRVGWHJPXO","download_json":"https://pith.science/pith/BO6O7YM6H75O5KSFKRVGWHJPXO.json","view_paper":"https://pith.science/paper/BO6O7YM6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1809.05096&json=true","fetch_graph":"https://pith.science/api/pith-number/BO6O7YM6H75O5KSFKRVGWHJPXO/graph.json","fetch_events":"https://pith.science/api/pith-number/BO6O7YM6H75O5KSFKRVGWHJPXO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BO6O7YM6H75O5KSFKRVGWHJPXO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BO6O7YM6H75O5KSFKRVGWHJPXO/action/storage_attestation","attest_author":"https://pith.science/pith/BO6O7YM6H75O5KSFKRVGWHJPXO/action/author_attestation","sign_citation":"https://pith.science/pith/BO6O7YM6H75O5KSFKRVGWHJPXO/action/citation_signature","submit_replication":"https://pith.science/pith/BO6O7YM6H75O5KSFKRVGWHJPXO/action/replication_record"}},"created_at":"2026-05-17T23:46:55.151835+00:00","updated_at":"2026-05-17T23:46:55.151835+00:00"}