{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NTZ6DD6SKFW4GL3SCDK4ZXAVEI","short_pith_number":"pith:NTZ6DD6S","schema_version":"1.0","canonical_sha256":"6cf3e18fd2516dc32f7210d5ccdc152214284571a069b707dc8b9edbd782dfee","source":{"kind":"arxiv","id":"2303.01768","version":1},"attestation_state":"computed","paper":{"title":"Toward Risk-based Optimistic Exploration for Cooperative Multi-Agent Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.LG","authors_text":"Jihwan Oh, Joonkee Kim, Minchan Jeong, Se-Young Yun","submitted_at":"2023-03-03T08:17:57Z","abstract_excerpt":"The multi-agent setting is intricate and unpredictable since the behaviors of multiple agents influence one another. To address this environmental uncertainty, distributional reinforcement learning algorithms that incorporate uncertainty via distributional output have been integrated with multi-agent reinforcement learning (MARL) methods, achieving state-of-the-art performance. However, distributional MARL algorithms still rely on the traditional $\\epsilon$-greedy, which does not take cooperative strategy into account. In this paper, we present a risk-based exploration that leads to collaborat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.01768","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-03-03T08:17:57Z","cross_cats_sorted":["cs.MA"],"title_canon_sha256":"1f3a9ef4dac4aeff40fd3c91aef85691e8fe16982d5a8d4b3094b1d82e7d0934","abstract_canon_sha256":"12ed57619607bc9dbac869dc8367cf757b2debf4689dfacc8d48ec451568dafd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:47:48.642045Z","signature_b64":"Lg7fV0qHXbkLoMEvgGyny0lW3dYjaUNSLLAMuyC9EmHoC9toxhVQtS0NiKUvZhCO6jZtFcGUbo7VNwP4CkmzCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6cf3e18fd2516dc32f7210d5ccdc152214284571a069b707dc8b9edbd782dfee","last_reissued_at":"2026-07-05T05:47:48.641542Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:47:48.641542Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Toward Risk-based Optimistic Exploration for Cooperative Multi-Agent Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.LG","authors_text":"Jihwan Oh, Joonkee Kim, Minchan Jeong, Se-Young Yun","submitted_at":"2023-03-03T08:17:57Z","abstract_excerpt":"The multi-agent setting is intricate and unpredictable since the behaviors of multiple agents influence one another. To address this environmental uncertainty, distributional reinforcement learning algorithms that incorporate uncertainty via distributional output have been integrated with multi-agent reinforcement learning (MARL) methods, achieving state-of-the-art performance. However, distributional MARL algorithms still rely on the traditional $\\epsilon$-greedy, which does not take cooperative strategy into account. In this paper, we present a risk-based exploration that leads to collaborat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.01768","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.01768/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.01768","created_at":"2026-07-05T05:47:48.641603+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.01768v1","created_at":"2026-07-05T05:47:48.641603+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.01768","created_at":"2026-07-05T05:47:48.641603+00:00"},{"alias_kind":"pith_short_12","alias_value":"NTZ6DD6SKFW4","created_at":"2026-07-05T05:47:48.641603+00:00"},{"alias_kind":"pith_short_16","alias_value":"NTZ6DD6SKFW4GL3S","created_at":"2026-07-05T05:47:48.641603+00:00"},{"alias_kind":"pith_short_8","alias_value":"NTZ6DD6S","created_at":"2026-07-05T05:47:48.641603+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.12061","citing_title":"Tackling Uncertainties in Multi-Agent Reinforcement Learning through Integration of Agent Termination Dynamics","ref_index":21,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NTZ6DD6SKFW4GL3SCDK4ZXAVEI","json":"https://pith.science/pith/NTZ6DD6SKFW4GL3SCDK4ZXAVEI.json","graph_json":"https://pith.science/api/pith-number/NTZ6DD6SKFW4GL3SCDK4ZXAVEI/graph.json","events_json":"https://pith.science/api/pith-number/NTZ6DD6SKFW4GL3SCDK4ZXAVEI/events.json","paper":"https://pith.science/paper/NTZ6DD6S"},"agent_actions":{"view_html":"https://pith.science/pith/NTZ6DD6SKFW4GL3SCDK4ZXAVEI","download_json":"https://pith.science/pith/NTZ6DD6SKFW4GL3SCDK4ZXAVEI.json","view_paper":"https://pith.science/paper/NTZ6DD6S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.01768&json=true","fetch_graph":"https://pith.science/api/pith-number/NTZ6DD6SKFW4GL3SCDK4ZXAVEI/graph.json","fetch_events":"https://pith.science/api/pith-number/NTZ6DD6SKFW4GL3SCDK4ZXAVEI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NTZ6DD6SKFW4GL3SCDK4ZXAVEI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NTZ6DD6SKFW4GL3SCDK4ZXAVEI/action/storage_attestation","attest_author":"https://pith.science/pith/NTZ6DD6SKFW4GL3SCDK4ZXAVEI/action/author_attestation","sign_citation":"https://pith.science/pith/NTZ6DD6SKFW4GL3SCDK4ZXAVEI/action/citation_signature","submit_replication":"https://pith.science/pith/NTZ6DD6SKFW4GL3SCDK4ZXAVEI/action/replication_record"}},"created_at":"2026-07-05T05:47:48.641603+00:00","updated_at":"2026-07-05T05:47:48.641603+00:00"}