{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:VZTQIUWAIDICSKA7WOVMTAYETX","short_pith_number":"pith:VZTQIUWA","schema_version":"1.0","canonical_sha256":"ae670452c040d029281fb3aac983049dc0466d82b9b4dc2eca3eca289364bb61","source":{"kind":"arxiv","id":"2006.14171","version":3},"attestation_state":"computed","paper":{"title":"A Closer Look at Invalid Action Masking in Policy Gradient Algorithms","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Santiago Onta\\~n\\'on, Shengyi Huang","submitted_at":"2020-06-25T04:47:09Z","abstract_excerpt":"In recent years, Deep Reinforcement Learning (DRL) algorithms have achieved state-of-the-art performance in many challenging strategy games. Because these games have complicated rules, an action sampled from the full discrete action distribution predicted by the learned policy is likely to be invalid according to the game rules (e.g., walking into a wall). The usual approach to deal with this problem in policy gradient algorithms is to \"mask out\" invalid actions and just sample from the set of valid actions. The implications of this process, however, remain under-investigated. In this paper, w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2006.14171","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-06-25T04:47:09Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"a358de2e6557031cf31d1a021fd6787c801f4cdcb0cdb1e7be2b34c7583d4d25","abstract_canon_sha256":"d2ae0698e88b7289499f4c3b1049e7b6e9900d556d48defec2d99c3ce2aa5c8e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:27:27.204222Z","signature_b64":"uedtAYtNur79OAPDZ38DaEVhjGXEgPTAFk01BcG2WNWpxT76CPaVSRloNPz5RI+eK6W2vKZpAigBm1Y9D7ggBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ae670452c040d029281fb3aac983049dc0466d82b9b4dc2eca3eca289364bb61","last_reissued_at":"2026-07-05T04:27:27.203730Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:27:27.203730Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Closer Look at Invalid Action Masking in Policy Gradient Algorithms","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Santiago Onta\\~n\\'on, Shengyi Huang","submitted_at":"2020-06-25T04:47:09Z","abstract_excerpt":"In recent years, Deep Reinforcement Learning (DRL) algorithms have achieved state-of-the-art performance in many challenging strategy games. Because these games have complicated rules, an action sampled from the full discrete action distribution predicted by the learned policy is likely to be invalid according to the game rules (e.g., walking into a wall). The usual approach to deal with this problem in policy gradient algorithms is to \"mask out\" invalid actions and just sample from the set of valid actions. The implications of this process, however, remain under-investigated. In this paper, w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2006.14171","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2006.14171/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2006.14171","created_at":"2026-07-05T04:27:27.203780+00:00"},{"alias_kind":"arxiv_version","alias_value":"2006.14171v3","created_at":"2026-07-05T04:27:27.203780+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2006.14171","created_at":"2026-07-05T04:27:27.203780+00:00"},{"alias_kind":"pith_short_12","alias_value":"VZTQIUWAIDIC","created_at":"2026-07-05T04:27:27.203780+00:00"},{"alias_kind":"pith_short_16","alias_value":"VZTQIUWAIDICSKA7","created_at":"2026-07-05T04:27:27.203780+00:00"},{"alias_kind":"pith_short_8","alias_value":"VZTQIUWA","created_at":"2026-07-05T04:27:27.203780+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22159","citing_title":"Deep RL for Fast Long-Horizon Operations Scheduling on NASA's Carruthers Geocorona Observatory Mission","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10979","citing_title":"Bellman-Taylor Score Decoding for Markov Decision Processes with State-Dependent Feasible Action Sets","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15236","citing_title":"Learning Selective Merge Policies for Deadline-Constrained Coded Caching via Deep Reinforcement Learning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28730","citing_title":"AlphaTransit: Learning to Design City-scale Transit Routes","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15236","citing_title":"Learning Selective Merge Policies for Deadline-Constrained Coded Caching via Deep Reinforcement Learning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11375","citing_title":"TuniQ: Autotuning Compilation Passes for Quantum Workloads at Scale for Effectiveness and Efficiency","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24501","citing_title":"TARMM: Scaling Delay-Critical Edge AI Offloading in 5G O-RAN via Temporal Graph Mobility Management","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01025","citing_title":"Your Loss is My Gain: Low Stake Attacks on Liquid Staking Pools","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VZTQIUWAIDICSKA7WOVMTAYETX","json":"https://pith.science/pith/VZTQIUWAIDICSKA7WOVMTAYETX.json","graph_json":"https://pith.science/api/pith-number/VZTQIUWAIDICSKA7WOVMTAYETX/graph.json","events_json":"https://pith.science/api/pith-number/VZTQIUWAIDICSKA7WOVMTAYETX/events.json","paper":"https://pith.science/paper/VZTQIUWA"},"agent_actions":{"view_html":"https://pith.science/pith/VZTQIUWAIDICSKA7WOVMTAYETX","download_json":"https://pith.science/pith/VZTQIUWAIDICSKA7WOVMTAYETX.json","view_paper":"https://pith.science/paper/VZTQIUWA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2006.14171&json=true","fetch_graph":"https://pith.science/api/pith-number/VZTQIUWAIDICSKA7WOVMTAYETX/graph.json","fetch_events":"https://pith.science/api/pith-number/VZTQIUWAIDICSKA7WOVMTAYETX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VZTQIUWAIDICSKA7WOVMTAYETX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VZTQIUWAIDICSKA7WOVMTAYETX/action/storage_attestation","attest_author":"https://pith.science/pith/VZTQIUWAIDICSKA7WOVMTAYETX/action/author_attestation","sign_citation":"https://pith.science/pith/VZTQIUWAIDICSKA7WOVMTAYETX/action/citation_signature","submit_replication":"https://pith.science/pith/VZTQIUWAIDICSKA7WOVMTAYETX/action/replication_record"}},"created_at":"2026-07-05T04:27:27.203780+00:00","updated_at":"2026-07-05T04:27:27.203780+00:00"}