{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FCEVXODYOUGURBHUUYA5OEZKZO","short_pith_number":"pith:FCEVXODY","schema_version":"1.0","canonical_sha256":"28895bb878750d4884f4a601d7132acb8998ff0ed1294b3878a0e36a697e7307","source":{"kind":"arxiv","id":"2301.05334","version":1},"attestation_state":"computed","paper":{"title":"TransfQMix: Transformers for Leveraging the Graph Structure of Multi-Agent Reinforcement Learning Problems","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.MA"],"primary_cat":"cs.LG","authors_text":"Ivan Masmitja, Mario Martin, Matteo Gallici","submitted_at":"2023-01-13T00:07:08Z","abstract_excerpt":"Coordination is one of the most difficult aspects of multi-agent reinforcement learning (MARL). One reason is that agents normally choose their actions independently of one another. In order to see coordination strategies emerging from the combination of independent policies, the recent research has focused on the use of a centralized function (CF) that learns each agent's contribution to the team reward. However, the structure in which the environment is presented to the agents and to the CF is typically overlooked. We have observed that the features used to describe the coordination problem "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.05334","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2023-01-13T00:07:08Z","cross_cats_sorted":["cs.AI","cs.MA"],"title_canon_sha256":"921ad800c58080e1e5bc4692a59d6cac041c9a794894d20ff9dc41d54e375b13","abstract_canon_sha256":"1db2c1109d1a1704ce87a559c20c9a07939b95dbe9227a9df68e48bb6ffce8d7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:32:51.886293Z","signature_b64":"bUXKe4iOIxXu99EpltE7QxnMlYRni+OuSJm6SSm3Xvbs+UNXomnDrIl6ZeLogkLBMlj+9hKMyMjxo3RVOZR0Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"28895bb878750d4884f4a601d7132acb8998ff0ed1294b3878a0e36a697e7307","last_reissued_at":"2026-07-05T05:32:51.885877Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:32:51.885877Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TransfQMix: Transformers for Leveraging the Graph Structure of Multi-Agent Reinforcement Learning Problems","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.MA"],"primary_cat":"cs.LG","authors_text":"Ivan Masmitja, Mario Martin, Matteo Gallici","submitted_at":"2023-01-13T00:07:08Z","abstract_excerpt":"Coordination is one of the most difficult aspects of multi-agent reinforcement learning (MARL). One reason is that agents normally choose their actions independently of one another. In order to see coordination strategies emerging from the combination of independent policies, the recent research has focused on the use of a centralized function (CF) that learns each agent's contribution to the team reward. However, the structure in which the environment is presented to the agents and to the CF is typically overlooked. We have observed that the features used to describe the coordination problem "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.05334","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.05334/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.05334","created_at":"2026-07-05T05:32:51.885931+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.05334v1","created_at":"2026-07-05T05:32:51.885931+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.05334","created_at":"2026-07-05T05:32:51.885931+00:00"},{"alias_kind":"pith_short_12","alias_value":"FCEVXODYOUGU","created_at":"2026-07-05T05:32:51.885931+00:00"},{"alias_kind":"pith_short_16","alias_value":"FCEVXODYOUGURBHU","created_at":"2026-07-05T05:32:51.885931+00:00"},{"alias_kind":"pith_short_8","alias_value":"FCEVXODY","created_at":"2026-07-05T05:32:51.885931+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05208","citing_title":"Transformer-Enhanced Reinforcement Learning: Fundamentals and Applications in Communication Networks","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2502.00558","citing_title":"Asynchronous Cooperative Multi-Agent Reinforcement Learning with Limited Communication","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FCEVXODYOUGURBHUUYA5OEZKZO","json":"https://pith.science/pith/FCEVXODYOUGURBHUUYA5OEZKZO.json","graph_json":"https://pith.science/api/pith-number/FCEVXODYOUGURBHUUYA5OEZKZO/graph.json","events_json":"https://pith.science/api/pith-number/FCEVXODYOUGURBHUUYA5OEZKZO/events.json","paper":"https://pith.science/paper/FCEVXODY"},"agent_actions":{"view_html":"https://pith.science/pith/FCEVXODYOUGURBHUUYA5OEZKZO","download_json":"https://pith.science/pith/FCEVXODYOUGURBHUUYA5OEZKZO.json","view_paper":"https://pith.science/paper/FCEVXODY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.05334&json=true","fetch_graph":"https://pith.science/api/pith-number/FCEVXODYOUGURBHUUYA5OEZKZO/graph.json","fetch_events":"https://pith.science/api/pith-number/FCEVXODYOUGURBHUUYA5OEZKZO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FCEVXODYOUGURBHUUYA5OEZKZO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FCEVXODYOUGURBHUUYA5OEZKZO/action/storage_attestation","attest_author":"https://pith.science/pith/FCEVXODYOUGURBHUUYA5OEZKZO/action/author_attestation","sign_citation":"https://pith.science/pith/FCEVXODYOUGURBHUUYA5OEZKZO/action/citation_signature","submit_replication":"https://pith.science/pith/FCEVXODYOUGURBHUUYA5OEZKZO/action/replication_record"}},"created_at":"2026-07-05T05:32:51.885931+00:00","updated_at":"2026-07-05T05:32:51.885931+00:00"}