{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:O2ATHKSCZGS73OUSYIBS6775RJ","short_pith_number":"pith:O2ATHKSC","schema_version":"1.0","canonical_sha256":"768133aa42c9a5fdba92c2032f7ffd8a7eb7d8dad22b6d53d95fa8c975f8b5e9","source":{"kind":"arxiv","id":"2508.13661","version":4},"attestation_state":"computed","paper":{"title":"Communication-Enhanced Tutoring for Efficient Decentralized Multi-Agent Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.LG","authors_text":"Bogusz Stefa\\'nczyk, Dominik Bogucki, {\\L}ukasz Lepak, Maciej Wojtala, Pawe{\\l} Wawrzy\\'nski","submitted_at":"2025-08-19T09:08:48Z","abstract_excerpt":"Centralized Training with Decentralized Execution (CTDE) is the dominant paradigm in multi-agent reinforcement learning (MARL), enabling agents to act independently at test time while leveraging additional information during training. However, the most prominent methods within CTDE, based on value decomposition, are limited in learning efficiency and final performance by partial observability in both training and execution. To overcome this limitation, in this work, we propose the framework of tutoring: In training, the agents share information in their latent space to develop well-informed po"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.13661","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-19T09:08:48Z","cross_cats_sorted":["cs.MA"],"title_canon_sha256":"8d126e10b5c61bf09c39521b13114d24a177fcabe653beac1d600d5a19c7dabc","abstract_canon_sha256":"bae64af6336089535f7fb8cd32e51a2b353665b81177a08fbe692235111afea1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-06T01:37:08.459572Z","signature_b64":"R7QMOWa87OzwlkCzk651UfMUhdiLeH7X+84EuGwYwT84qH4T7HPssV4ymbqILEKPJALc6eWJNNObK67GQwO1CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"768133aa42c9a5fdba92c2032f7ffd8a7eb7d8dad22b6d53d95fa8c975f8b5e9","last_reissued_at":"2026-08-06T01:37:08.457963Z","signature_status":"signed_v1","first_computed_at":"2026-08-06T01:37:08.457963Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Communication-Enhanced Tutoring for Efficient Decentralized Multi-Agent Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.LG","authors_text":"Bogusz Stefa\\'nczyk, Dominik Bogucki, {\\L}ukasz Lepak, Maciej Wojtala, Pawe{\\l} Wawrzy\\'nski","submitted_at":"2025-08-19T09:08:48Z","abstract_excerpt":"Centralized Training with Decentralized Execution (CTDE) is the dominant paradigm in multi-agent reinforcement learning (MARL), enabling agents to act independently at test time while leveraging additional information during training. However, the most prominent methods within CTDE, based on value decomposition, are limited in learning efficiency and final performance by partial observability in both training and execution. To overcome this limitation, in this work, we propose the framework of tutoring: In training, the agents share information in their latent space to develop well-informed po"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.13661","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.13661/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.13661","created_at":"2026-08-06T01:37:08.459779+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.13661v4","created_at":"2026-08-06T01:37:08.459779+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.13661","created_at":"2026-08-06T01:37:08.459779+00:00"},{"alias_kind":"pith_short_12","alias_value":"O2ATHKSCZGS7","created_at":"2026-08-06T01:37:08.459779+00:00"},{"alias_kind":"pith_short_16","alias_value":"O2ATHKSCZGS73OUS","created_at":"2026-08-06T01:37:08.459779+00:00"},{"alias_kind":"pith_short_8","alias_value":"O2ATHKSC","created_at":"2026-08-06T01:37:08.459779+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2606.11249","citing_title":"MASK: Multi-Agent Semantic K-Scheduling for Risk-Sensitive 6G Robotics","ref_index":16,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O2ATHKSCZGS73OUSYIBS6775RJ","json":"https://pith.science/pith/O2ATHKSCZGS73OUSYIBS6775RJ.json","graph_json":"https://pith.science/api/pith-number/O2ATHKSCZGS73OUSYIBS6775RJ/graph.json","events_json":"https://pith.science/api/pith-number/O2ATHKSCZGS73OUSYIBS6775RJ/events.json","paper":"https://pith.science/paper/O2ATHKSC"},"agent_actions":{"view_html":"https://pith.science/pith/O2ATHKSCZGS73OUSYIBS6775RJ","download_json":"https://pith.science/pith/O2ATHKSCZGS73OUSYIBS6775RJ.json","view_paper":"https://pith.science/paper/O2ATHKSC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.13661&json=true","fetch_graph":"https://pith.science/api/pith-number/O2ATHKSCZGS73OUSYIBS6775RJ/graph.json","fetch_events":"https://pith.science/api/pith-number/O2ATHKSCZGS73OUSYIBS6775RJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O2ATHKSCZGS73OUSYIBS6775RJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O2ATHKSCZGS73OUSYIBS6775RJ/action/storage_attestation","attest_author":"https://pith.science/pith/O2ATHKSCZGS73OUSYIBS6775RJ/action/author_attestation","sign_citation":"https://pith.science/pith/O2ATHKSCZGS73OUSYIBS6775RJ/action/citation_signature","submit_replication":"https://pith.science/pith/O2ATHKSCZGS73OUSYIBS6775RJ/action/replication_record"}},"created_at":"2026-08-06T01:37:08.459779+00:00","updated_at":"2026-08-06T01:37:08.459779+00:00"}