{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:EZ5UJQM5TMKUZ575AU2EZJISAS","short_pith_number":"pith:EZ5UJQM5","schema_version":"1.0","canonical_sha256":"267b44c19d9b154cf7fd05344ca51204a578ad181222d8daec5b09b3738ba38e","source":{"kind":"arxiv","id":"2112.11731","version":1},"attestation_state":"computed","paper":{"title":"Graph augmented Deep Reinforcement Learning in the GameRLand3D environment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Christian Wolf, Edward Beeching, Jilles Debangoye, Joshua Romoff, Maxim Peter, Olivier Simonin, Philippe Marcotte","submitted_at":"2021-12-22T08:48:00Z","abstract_excerpt":"We address planning and navigation in challenging 3D video games featuring maps with disconnected regions reachable by agents using special actions. In this setting, classical symbolic planners are not applicable or difficult to adapt. We introduce a hybrid technique combining a low level policy trained with reinforcement learning and a graph based high level classical planner. In addition to providing human-interpretable paths, the approach improves the generalization performance of an end-to-end approach in unseen maps, where it achieves a 20% absolute increase in success rate over a recurre"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.11731","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-12-22T08:48:00Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"dd9c22f474b86fcd86023026a63607e488c9daa9eaf42739d4468616590f3e4d","abstract_canon_sha256":"428fc1ec3ba8043615b0cb85b8ccff2f3a3e019f04b298b56ec027828d1b60bd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:43:10.386637Z","signature_b64":"bRv5ROA7JEB6zw3ib2mE6kWrxp+Of5Go+rFHkFu+C0ueY5GUqFJBgoBoB8U6UqDnvzBRruNYX6qJJkzS5bbMAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"267b44c19d9b154cf7fd05344ca51204a578ad181222d8daec5b09b3738ba38e","last_reissued_at":"2026-07-05T03:43:10.386186Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:43:10.386186Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Graph augmented Deep Reinforcement Learning in the GameRLand3D environment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Christian Wolf, Edward Beeching, Jilles Debangoye, Joshua Romoff, Maxim Peter, Olivier Simonin, Philippe Marcotte","submitted_at":"2021-12-22T08:48:00Z","abstract_excerpt":"We address planning and navigation in challenging 3D video games featuring maps with disconnected regions reachable by agents using special actions. In this setting, classical symbolic planners are not applicable or difficult to adapt. We introduce a hybrid technique combining a low level policy trained with reinforcement learning and a graph based high level classical planner. In addition to providing human-interpretable paths, the approach improves the generalization performance of an end-to-end approach in unseen maps, where it achieves a 20% absolute increase in success rate over a recurre"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.11731","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.11731/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.11731","created_at":"2026-07-05T03:43:10.386242+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.11731v1","created_at":"2026-07-05T03:43:10.386242+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.11731","created_at":"2026-07-05T03:43:10.386242+00:00"},{"alias_kind":"pith_short_12","alias_value":"EZ5UJQM5TMKU","created_at":"2026-07-05T03:43:10.386242+00:00"},{"alias_kind":"pith_short_16","alias_value":"EZ5UJQM5TMKUZ575","created_at":"2026-07-05T03:43:10.386242+00:00"},{"alias_kind":"pith_short_8","alias_value":"EZ5UJQM5","created_at":"2026-07-05T03:43:10.386242+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.07177","citing_title":"Effective Reward Specification in Deep Reinforcement Learning","ref_index":35,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EZ5UJQM5TMKUZ575AU2EZJISAS","json":"https://pith.science/pith/EZ5UJQM5TMKUZ575AU2EZJISAS.json","graph_json":"https://pith.science/api/pith-number/EZ5UJQM5TMKUZ575AU2EZJISAS/graph.json","events_json":"https://pith.science/api/pith-number/EZ5UJQM5TMKUZ575AU2EZJISAS/events.json","paper":"https://pith.science/paper/EZ5UJQM5"},"agent_actions":{"view_html":"https://pith.science/pith/EZ5UJQM5TMKUZ575AU2EZJISAS","download_json":"https://pith.science/pith/EZ5UJQM5TMKUZ575AU2EZJISAS.json","view_paper":"https://pith.science/paper/EZ5UJQM5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.11731&json=true","fetch_graph":"https://pith.science/api/pith-number/EZ5UJQM5TMKUZ575AU2EZJISAS/graph.json","fetch_events":"https://pith.science/api/pith-number/EZ5UJQM5TMKUZ575AU2EZJISAS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EZ5UJQM5TMKUZ575AU2EZJISAS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EZ5UJQM5TMKUZ575AU2EZJISAS/action/storage_attestation","attest_author":"https://pith.science/pith/EZ5UJQM5TMKUZ575AU2EZJISAS/action/author_attestation","sign_citation":"https://pith.science/pith/EZ5UJQM5TMKUZ575AU2EZJISAS/action/citation_signature","submit_replication":"https://pith.science/pith/EZ5UJQM5TMKUZ575AU2EZJISAS/action/replication_record"}},"created_at":"2026-07-05T03:43:10.386242+00:00","updated_at":"2026-07-05T03:43:10.386242+00:00"}