{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:7GOVFDXB2CNWFYXOJTA75DICSC","short_pith_number":"pith:7GOVFDXB","schema_version":"1.0","canonical_sha256":"f99d528ee1d09b62e2ee4cc1fe8d02908b095d6801ca35c084dce8f292c16509","source":{"kind":"arxiv","id":"2201.04612","version":1},"attestation_state":"computed","paper":{"title":"Agent-Temporal Attention for Reward Redistribution in Episodic Multi-Agent Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.MA","authors_text":"Baicen Xiao, Bhaskar Ramasubramanian, Radha Poovendran","submitted_at":"2022-01-12T18:35:46Z","abstract_excerpt":"This paper considers multi-agent reinforcement learning (MARL) tasks where agents receive a shared global reward at the end of an episode. The delayed nature of this reward affects the ability of the agents to assess the quality of their actions at intermediate time-steps. This paper focuses on developing methods to learn a temporal redistribution of the episodic reward to obtain a dense reward signal. Solving such MARL problems requires addressing two challenges: identifying (1) relative importance of states along the length of an episode (along time), and (2) relative importance of individua"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2201.04612","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.MA","submitted_at":"2022-01-12T18:35:46Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"682d3b783b1802cdf9ea1c6f1885097f6d90ce7dd21b197951f128247ba5a21f","abstract_canon_sha256":"bda64f336d5384bb2ec070298940f7560ec003215b787395cf9740eddcd7fa24"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:48:02.053026Z","signature_b64":"yJ/EreX/w7LhKxnZQfrO3rcWzBuayIt0w6AGxWa+YtJzXQj8Ly6QindWQ4gDEno7hIdwse4nVTBHvfwl9oLgBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f99d528ee1d09b62e2ee4cc1fe8d02908b095d6801ca35c084dce8f292c16509","last_reissued_at":"2026-07-05T03:48:02.052519Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:48:02.052519Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Agent-Temporal Attention for Reward Redistribution in Episodic Multi-Agent Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.MA","authors_text":"Baicen Xiao, Bhaskar Ramasubramanian, Radha Poovendran","submitted_at":"2022-01-12T18:35:46Z","abstract_excerpt":"This paper considers multi-agent reinforcement learning (MARL) tasks where agents receive a shared global reward at the end of an episode. The delayed nature of this reward affects the ability of the agents to assess the quality of their actions at intermediate time-steps. This paper focuses on developing methods to learn a temporal redistribution of the episodic reward to obtain a dense reward signal. Solving such MARL problems requires addressing two challenges: identifying (1) relative importance of states along the length of an episode (along time), and (2) relative importance of individua"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2201.04612","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2201.04612/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2201.04612","created_at":"2026-07-05T03:48:02.052592+00:00"},{"alias_kind":"arxiv_version","alias_value":"2201.04612v1","created_at":"2026-07-05T03:48:02.052592+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2201.04612","created_at":"2026-07-05T03:48:02.052592+00:00"},{"alias_kind":"pith_short_12","alias_value":"7GOVFDXB2CNW","created_at":"2026-07-05T03:48:02.052592+00:00"},{"alias_kind":"pith_short_16","alias_value":"7GOVFDXB2CNWFYXO","created_at":"2026-07-05T03:48:02.052592+00:00"},{"alias_kind":"pith_short_8","alias_value":"7GOVFDXB","created_at":"2026-07-05T03:48:02.052592+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30246","citing_title":"Clarus: Coordinating Autonomous Research Agents toward Web-Scale Scientific Collaboration","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7GOVFDXB2CNWFYXOJTA75DICSC","json":"https://pith.science/pith/7GOVFDXB2CNWFYXOJTA75DICSC.json","graph_json":"https://pith.science/api/pith-number/7GOVFDXB2CNWFYXOJTA75DICSC/graph.json","events_json":"https://pith.science/api/pith-number/7GOVFDXB2CNWFYXOJTA75DICSC/events.json","paper":"https://pith.science/paper/7GOVFDXB"},"agent_actions":{"view_html":"https://pith.science/pith/7GOVFDXB2CNWFYXOJTA75DICSC","download_json":"https://pith.science/pith/7GOVFDXB2CNWFYXOJTA75DICSC.json","view_paper":"https://pith.science/paper/7GOVFDXB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2201.04612&json=true","fetch_graph":"https://pith.science/api/pith-number/7GOVFDXB2CNWFYXOJTA75DICSC/graph.json","fetch_events":"https://pith.science/api/pith-number/7GOVFDXB2CNWFYXOJTA75DICSC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7GOVFDXB2CNWFYXOJTA75DICSC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7GOVFDXB2CNWFYXOJTA75DICSC/action/storage_attestation","attest_author":"https://pith.science/pith/7GOVFDXB2CNWFYXOJTA75DICSC/action/author_attestation","sign_citation":"https://pith.science/pith/7GOVFDXB2CNWFYXOJTA75DICSC/action/citation_signature","submit_replication":"https://pith.science/pith/7GOVFDXB2CNWFYXOJTA75DICSC/action/replication_record"}},"created_at":"2026-07-05T03:48:02.052592+00:00","updated_at":"2026-07-05T03:48:02.052592+00:00"}