{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:Q7GPFNCV5IZQYNMVOVSTZ3DD7S","short_pith_number":"pith:Q7GPFNCV","schema_version":"1.0","canonical_sha256":"87ccf2b455ea330c359575653cec63fc959d5d23a97d339d17d5f3afb4a65365","source":{"kind":"arxiv","id":"1903.08082","version":1},"attestation_state":"computed","paper":{"title":"Learning Reciprocity in Complex Sequential Social Dilemmas","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.MA","authors_text":"Edward Hughes, J\\'anos Kram\\'ar, Joel Z. Leibo, Steven Wheelwright, Tom Eccles","submitted_at":"2019-03-19T16:18:00Z","abstract_excerpt":"Reciprocity is an important feature of human social interaction and underpins our cooperative nature. What is more, simple forms of reciprocity have proved remarkably resilient in matrix game social dilemmas. Most famously, the tit-for-tat strategy performs very well in tournaments of Prisoner's Dilemma. Unfortunately this strategy is not readily applicable to the real world, in which options to cooperate or defect are temporally and spatially extended. Here, we present a general online reinforcement learning algorithm that displays reciprocal behavior towards its co-players. We show that it c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1903.08082","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.MA","submitted_at":"2019-03-19T16:18:00Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"713fa0bd84b92544f8bb3f652425ea0f902fed30b11119e308b07340bbd7eed8","abstract_canon_sha256":"171baf961ccb60e2bf81c1654abe818be81353cf09d6cc51b0548d3b90026e9b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-17T23:50:52.971169Z","signature_b64":"HyQVeHTVIJ2imB+laIico8VG0Nttl5RdmfUsolSrYU6bULB2p3Jef7+OTdSgpuYGuQtO9GnTV5KIOCObeTg0Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"87ccf2b455ea330c359575653cec63fc959d5d23a97d339d17d5f3afb4a65365","last_reissued_at":"2026-05-17T23:50:52.970446Z","signature_status":"signed_v1","first_computed_at":"2026-05-17T23:50:52.970446Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Reciprocity in Complex Sequential Social Dilemmas","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.MA","authors_text":"Edward Hughes, J\\'anos Kram\\'ar, Joel Z. Leibo, Steven Wheelwright, Tom Eccles","submitted_at":"2019-03-19T16:18:00Z","abstract_excerpt":"Reciprocity is an important feature of human social interaction and underpins our cooperative nature. What is more, simple forms of reciprocity have proved remarkably resilient in matrix game social dilemmas. Most famously, the tit-for-tat strategy performs very well in tournaments of Prisoner's Dilemma. Unfortunately this strategy is not readily applicable to the real world, in which options to cooperate or defect are temporally and spatially extended. Here, we present a general online reinforcement learning algorithm that displays reciprocal behavior towards its co-players. We show that it c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1903.08082","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1903.08082","created_at":"2026-05-17T23:50:52.970558+00:00"},{"alias_kind":"arxiv_version","alias_value":"1903.08082v1","created_at":"2026-05-17T23:50:52.970558+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1903.08082","created_at":"2026-05-17T23:50:52.970558+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q7GPFNCV5IZQ","created_at":"2026-05-18T12:33:27.125529+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q7GPFNCV5IZQYNMV","created_at":"2026-05-18T12:33:27.125529+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q7GPFNCV","created_at":"2026-05-18T12:33:27.125529+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q7GPFNCV5IZQYNMVOVSTZ3DD7S","json":"https://pith.science/pith/Q7GPFNCV5IZQYNMVOVSTZ3DD7S.json","graph_json":"https://pith.science/api/pith-number/Q7GPFNCV5IZQYNMVOVSTZ3DD7S/graph.json","events_json":"https://pith.science/api/pith-number/Q7GPFNCV5IZQYNMVOVSTZ3DD7S/events.json","paper":"https://pith.science/paper/Q7GPFNCV"},"agent_actions":{"view_html":"https://pith.science/pith/Q7GPFNCV5IZQYNMVOVSTZ3DD7S","download_json":"https://pith.science/pith/Q7GPFNCV5IZQYNMVOVSTZ3DD7S.json","view_paper":"https://pith.science/paper/Q7GPFNCV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1903.08082&json=true","fetch_graph":"https://pith.science/api/pith-number/Q7GPFNCV5IZQYNMVOVSTZ3DD7S/graph.json","fetch_events":"https://pith.science/api/pith-number/Q7GPFNCV5IZQYNMVOVSTZ3DD7S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q7GPFNCV5IZQYNMVOVSTZ3DD7S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q7GPFNCV5IZQYNMVOVSTZ3DD7S/action/storage_attestation","attest_author":"https://pith.science/pith/Q7GPFNCV5IZQYNMVOVSTZ3DD7S/action/author_attestation","sign_citation":"https://pith.science/pith/Q7GPFNCV5IZQYNMVOVSTZ3DD7S/action/citation_signature","submit_replication":"https://pith.science/pith/Q7GPFNCV5IZQYNMVOVSTZ3DD7S/action/replication_record"}},"created_at":"2026-05-17T23:50:52.970558+00:00","updated_at":"2026-05-17T23:50:52.970558+00:00"}