{"as_of":"2026-08-11T21:00:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0646529699c06dcf2d8995f31b83b5b65da5464812ddb51d10a181d232f21e5b","coverage":[{"denominator":15,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":15,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T09:01:03.535066Z","state":"measured"},{"denominator":15,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":15,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.18296/citation-record","integrity":"/paper/2607.18296/integrity","json":"/paper/2607.18296/citation-record.json","paper":"/paper/2607.18296"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T09:01:03.169475Z","title":"Baghchal: An augmented Q-learning approach to turn-based heterogeneous multi- agent systems,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.18296","last_updated":"2026-07-02T14:22:42Z","snapshot_observed_at":"2026-08-11T12:44:12.346679Z","submitted_at":"2026-07-02T14:22:42Z","title":"Deep Reinforcement Learning to Master the Asymmetric Strategy of Baghchal","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-02T09:01:03.169475Z"},"links":{"citing_paper":"/paper/2607.18296"},"observation_digest":"sha256:62e1e5a466de69dd62c257edc3cdd97e5410c9e43e9d7ebb7ea77c4e33ce73ac","observation_id":"a4e03f3c-509b-43d9-a625-c59a7173940a","resolution":{"observed_at":"2026-08-02T09:01:03.169475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T09:01:03.256634Z","title":"Human-level control through deep reinforcement learning,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2607.18296","last_updated":"2026-07-02T14:22:42Z","snapshot_observed_at":"2026-08-11T12:44:12.346679Z","submitted_at":"2026-07-02T14:22:42Z","title":"Deep Reinforcement Learning to Master the Asymmetric Strategy of Baghchal","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-02T09:01:03.256634Z"},"links":{"citing_paper":"/paper/2607.18296"},"observation_digest":"sha256:095fea7c5cb02bf10a3d2a17fe4f241468c76320be4e277e830f680d0a89d0ec","observation_id":"fb7b9925-d60f-4198-b1b5-bb12294043a1","resolution":{"observed_at":"2026-08-02T09:01:03.256634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T09:01:03.408707Z","title":"Simple statistical gradient-following algorithms for connectionist reinforcement learning,","venue":null,"work_id":null,"year":1992},"citing_paper":{"arxiv_id":"2607.18296","last_updated":"2026-07-02T14:22:42Z","snapshot_observed_at":"2026-08-11T12:44:12.346679Z","submitted_at":"2026-07-02T14:22:42Z","title":"Deep Reinforcement Learning to Master the Asymmetric Strategy of Baghchal","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-02T09:01:03.408707Z"},"links":{"citing_paper":"/paper/2607.18296"},"observation_digest":"sha256:9bb005f9d94728f2e7f76c3f412fa38e75630fda1a47cd00a088f0b037418c0c","observation_id":"56e1f317-fc36-4134-96e2-3102e50fea8a","resolution":{"observed_at":"2026-08-02T09:01:03.408707Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-02T09:01:03.483968Z","title":"Proximal policy optimization algorithms,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.18296","last_updated":"2026-07-02T14:22:42Z","snapshot_observed_at":"2026-08-11T12:44:12.346679Z","submitted_at":"2026-07-02T14:22:42Z","title":"Deep Reinforcement Learning to Master the Asymmetric Strategy of Baghchal","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-02T09:01:03.483968Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2607.18296"},"observation_digest":"sha256:13617b552ce6280558695e4f96bc4b8d8fd40d7fdaefec3b52c72c8df9ac1283","observation_id":"965f6a8d-d1ee-441a-a2f9-a9e610c5161e","resolution":{"observed_at":"2026-08-02T09:01:03.483968Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T09:01:03.489475Z","title":"MasteringAtari,Go,chessandshogibyplanningwith alearnedmodel,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.18296","last_updated":"2026-07-02T14:22:42Z","snapshot_observed_at":"2026-08-11T12:44:12.346679Z","submitted_at":"2026-07-02T14:22:42Z","title":"Deep Reinforcement Learning to Master the Asymmetric Strategy of Baghchal","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-02T09:01:03.489475Z"},"links":{"citing_paper":"/paper/2607.18296"},"observation_digest":"sha256:9021812554022edefdd6bee8e371431ddab6db4f0c30d49cafe3b92f027957ff","observation_id":"30654ff6-58d7-4413-92de-83b31636efea","resolution":{"observed_at":"2026-08-02T09:01:03.489475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T09:01:03.493918Z","title":"Mastering the game of Go with deep neural networks and treesearch,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2607.18296","last_updated":"2026-07-02T14:22:42Z","snapshot_observed_at":"2026-08-11T12:44:12.346679Z","submitted_at":"2026-07-02T14:22:42Z","title":"Deep Reinforcement Learning to Master the Asymmetric Strategy of Baghchal","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-02T09:01:03.493918Z"},"links":{"citing_paper":"/paper/2607.18296"},"observation_digest":"sha256:4103c46c3104f4f5af52a5e3b0a5c9fd2b265d578822e4cd2d1f9c28f9fd18fc","observation_id":"66dd54fe-b38f-4040-b336-7238dcea4e90","resolution":{"observed_at":"2026-08-02T09:01:03.493918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T09:01:03.499141Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.18296","last_updated":"2026-07-02T14:22:42Z","snapshot_observed_at":"2026-08-11T12:44:12.346679Z","submitted_at":"2026-07-02T14:22:42Z","title":"Deep Reinforcement Learning to Master the Asymmetric Strategy of Baghchal","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-02T09:01:03.499141Z"},"links":{"citing_paper":"/paper/2607.18296"},"observation_digest":"sha256:bdcc7e9bb57fdfe1a1a1fd25abbe82466dbd0115e0a14364aaa236c5587a6bfc","observation_id":"54404250-2567-450e-b218-4deaf81ac426","resolution":{"observed_at":"2026-08-02T09:01:03.499141Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T09:01:03.503169Z","title":"Superhuman AI for multiplayer poker,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.18296","last_updated":"2026-07-02T14:22:42Z","snapshot_observed_at":"2026-08-11T12:44:12.346679Z","submitted_at":"2026-07-02T14:22:42Z","title":"Deep Reinforcement Learning to Master the Asymmetric Strategy of Baghchal","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-02T09:01:03.503169Z"},"links":{"citing_paper":"/paper/2607.18296"},"observation_digest":"sha256:9d1295d8961269cb26b24dace124bcd14bb4d274647d35a293329e923b9abca4","observation_id":"28ae733c-5f2f-4285-93f8-7648c1a7625b","resolution":{"observed_at":"2026-08-02T09:01:03.503169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.01495","last_updated":"2018-02-23T10:04:20Z","snapshot_observed_at":"2026-08-05T02:06:43.683268Z","submitted_at":"2017-07-05T17:55:53Z","title":"Hindsight Experience Replay","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.01495","snapshot_observed_at":"2026-08-02T09:01:03.507472Z","title":"Multiagent deep reinforcement learning with extremelysparserewards,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.18296","last_updated":"2026-07-02T14:22:42Z","snapshot_observed_at":"2026-08-11T12:44:12.346679Z","submitted_at":"2026-07-02T14:22:42Z","title":"Deep Reinforcement Learning to Master the Asymmetric Strategy of Baghchal","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-02T09:01:03.507472Z"},"links":{"cited_paper":"/paper/1707.01495","citing_paper":"/paper/2607.18296"},"observation_digest":"sha256:7e428fa6f859f2a0797defdab65caf2cb20b7265ff736acd28592e2c668caf5b","observation_id":"8ed4ef66-14e5-4a09-adf2-8bbb11385577","resolution":{"observed_at":"2026-08-02T09:01:03.507472Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T09:01:03.512299Z","title":"Counterfactual multi-agent policy gradients,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2607.18296","last_updated":"2026-07-02T14:22:42Z","snapshot_observed_at":"2026-08-11T12:44:12.346679Z","submitted_at":"2026-07-02T14:22:42Z","title":"Deep Reinforcement Learning to Master the Asymmetric Strategy of Baghchal","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-02T09:01:03.512299Z"},"links":{"citing_paper":"/paper/2607.18296"},"observation_digest":"sha256:b74715697eb02e77d6b184fd3175ff8923613b7dbfafc14b9d78ae0810cefe22","observation_id":"1e5b21c0-fd39-4720-9d6d-5d07b6f3e568","resolution":{"observed_at":"2026-08-02T09:01:03.512299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T09:01:03.516477Z","title":"A unified game-theoretic approach to multiagent reinforcement learning,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.18296","last_updated":"2026-07-02T14:22:42Z","snapshot_observed_at":"2026-08-11T12:44:12.346679Z","submitted_at":"2026-07-02T14:22:42Z","title":"Deep Reinforcement Learning to Master the Asymmetric Strategy of Baghchal","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-02T09:01:03.516477Z"},"links":{"citing_paper":"/paper/2607.18296"},"observation_digest":"sha256:427aef33a33ba4d127a18173cc4b769367377e40cab76be7c5833444b0458ce6","observation_id":"89e3c950-78de-4a04-a9cd-0db7cb6a13c9","resolution":{"observed_at":"2026-08-02T09:01:03.516477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1906.11374","last_updated":"2024-01-02T18:15:23Z","snapshot_observed_at":"2026-08-06T15:37:44.419422Z","submitted_at":"2019-06-26T22:50:41Z","title":"Geometric phase predicts locomotion performance in undulating living systems across scales","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1906.11374","snapshot_observed_at":"2026-08-02T09:01:03.521396Z","title":"Dealing with non-stationarity in multi-agent deep reinforcement learning,","venue":null,"work_id":null,"year":1906},"citing_paper":{"arxiv_id":"2607.18296","last_updated":"2026-07-02T14:22:42Z","snapshot_observed_at":"2026-08-11T12:44:12.346679Z","submitted_at":"2026-07-02T14:22:42Z","title":"Deep Reinforcement Learning to Master the Asymmetric Strategy of Baghchal","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-02T09:01:03.521396Z"},"links":{"cited_paper":"/paper/1906.11374","citing_paper":"/paper/2607.18296"},"observation_digest":"sha256:18f91ac8991a6f98a5eebd74b3dc907f0ae5f97f15177016b92b9493d4d64a2d","observation_id":"81a406b0-0086-41ba-886f-8ea38382650a","resolution":{"observed_at":"2026-08-02T09:01:03.521396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T09:01:03.526243Z","title":"Deep Blue,","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2607.18296","last_updated":"2026-07-02T14:22:42Z","snapshot_observed_at":"2026-08-11T12:44:12.346679Z","submitted_at":"2026-07-02T14:22:42Z","title":"Deep Reinforcement Learning to Master the Asymmetric Strategy of Baghchal","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-02T09:01:03.526243Z"},"links":{"citing_paper":"/paper/2607.18296"},"observation_digest":"sha256:f7df2265f983b0b35b8c7d497e37fd4e3ef37d0c4fda85bf40b85f43529911cb","observation_id":"b1ee4d99-8911-4051-ad51-a6510ab2b416","resolution":{"observed_at":"2026-08-02T09:01:03.526243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1912.06680","last_updated":"2019-12-13T19:56:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-12-13T19:56:40Z","title":"Dota 2 with Large Scale Deep Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1912.06680","snapshot_observed_at":"2026-08-02T09:01:03.530354Z","title":"Dota 2 with large scale deep reinforcementlearning,","venue":null,"work_id":null,"year":1912},"citing_paper":{"arxiv_id":"2607.18296","last_updated":"2026-07-02T14:22:42Z","snapshot_observed_at":"2026-08-11T12:44:12.346679Z","submitted_at":"2026-07-02T14:22:42Z","title":"Deep Reinforcement Learning to Master the Asymmetric Strategy of Baghchal","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-02T09:01:03.530354Z"},"links":{"cited_paper":"/paper/1912.06680","citing_paper":"/paper/2607.18296"},"observation_digest":"sha256:9a781fd34b5e2a14c2a75421bd7980b7adaa5d6e9363b9a959586554fc0ac711","observation_id":"9d52b22d-868a-4dd0-8169-d4af2b2b0829","resolution":{"observed_at":"2026-08-02T09:01:03.530354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T09:01:03.535066Z","title":"AIstrategyapproachdevelopment on Baghchal using AlphaZero,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.18296","last_updated":"2026-07-02T14:22:42Z","snapshot_observed_at":"2026-08-11T12:44:12.346679Z","submitted_at":"2026-07-02T14:22:42Z","title":"Deep Reinforcement Learning to Master the Asymmetric Strategy of Baghchal","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-02T09:01:03.535066Z"},"links":{"citing_paper":"/paper/2607.18296"},"observation_digest":"sha256:869d32e2d99af9daa5b2bb9a7ad322c4424ff6dd7e2597e4e83e82f5f6412a85","observation_id":"dfa83d99-ec54-4cfa-9caf-4fb3e24fed20","resolution":{"observed_at":"2026-08-02T09:01:03.535066Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.18296","last_updated":"2026-07-02T14:22:42Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-11T12:44:12.346679Z","submitted_at":"2026-07-02T14:22:42Z","title":"Deep Reinforcement Learning to Master the Asymmetric Strategy of Baghchal"},"reference_resolution":{"displayed":15,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":15,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":15},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 15 of 15 outbound references and 0 inbound Pith citation observations for arXiv:2607.18296."}