{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:M4VP6FXTEWGC6UOTY3A3V5SESS","short_pith_number":"pith:M4VP6FXT","schema_version":"1.0","canonical_sha256":"672aff16f3258c2f51d3c6c1baf6449487c6da89b9a4e081939d5afdba8ed7c9","source":{"kind":"arxiv","id":"2311.15341","version":1},"attestation_state":"computed","paper":{"title":"Generative Modelling of Stochastic Actions with Arbitrary Constraints in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Arunesh Sinha, Changyu Chen, Pradeep Varakantham, Ramesha Karunasena, Thanh Hong Nguyen","submitted_at":"2023-11-26T15:57:20Z","abstract_excerpt":"Many problems in Reinforcement Learning (RL) seek an optimal policy with large discrete multidimensional yet unordered action spaces; these include problems in randomized allocation of resources such as placements of multiple security resources and emergency response units, etc. A challenge in this setting is that the underlying action space is categorical (discrete and unordered) and large, for which existing RL methods do not perform well. Moreover, these problems require validity of the realized action (allocation); this validity constraint is often difficult to express compactly in a close"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.15341","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-11-26T15:57:20Z","cross_cats_sorted":[],"title_canon_sha256":"4e929b169bc4bc845bc4f7811c9fd1abe974dc5944e2271a5095d6aa36788f82","abstract_canon_sha256":"a13ac19848b77c62f36baf4d30dd2c42f4abaecebeb94f45dcbaee084692c661"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:17:02.034641Z","signature_b64":"LMLnLfRZmM21SwuzTAZXMNPpeRg9T4J8l0DY/Dbkkigsw/d/rzuXWi5OX/B1dBJYsVoX7WqCQuMPFId353DTDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"672aff16f3258c2f51d3c6c1baf6449487c6da89b9a4e081939d5afdba8ed7c9","last_reissued_at":"2026-07-05T07:17:02.034229Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:17:02.034229Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generative Modelling of Stochastic Actions with Arbitrary Constraints in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Arunesh Sinha, Changyu Chen, Pradeep Varakantham, Ramesha Karunasena, Thanh Hong Nguyen","submitted_at":"2023-11-26T15:57:20Z","abstract_excerpt":"Many problems in Reinforcement Learning (RL) seek an optimal policy with large discrete multidimensional yet unordered action spaces; these include problems in randomized allocation of resources such as placements of multiple security resources and emergency response units, etc. A challenge in this setting is that the underlying action space is categorical (discrete and unordered) and large, for which existing RL methods do not perform well. Moreover, these problems require validity of the realized action (allocation); this validity constraint is often difficult to express compactly in a close"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.15341","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.15341/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.15341","created_at":"2026-07-05T07:17:02.034285+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.15341v1","created_at":"2026-07-05T07:17:02.034285+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.15341","created_at":"2026-07-05T07:17:02.034285+00:00"},{"alias_kind":"pith_short_12","alias_value":"M4VP6FXTEWGC","created_at":"2026-07-05T07:17:02.034285+00:00"},{"alias_kind":"pith_short_16","alias_value":"M4VP6FXTEWGC6UOT","created_at":"2026-07-05T07:17:02.034285+00:00"},{"alias_kind":"pith_short_8","alias_value":"M4VP6FXT","created_at":"2026-07-05T07:17:02.034285+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M4VP6FXTEWGC6UOTY3A3V5SESS","json":"https://pith.science/pith/M4VP6FXTEWGC6UOTY3A3V5SESS.json","graph_json":"https://pith.science/api/pith-number/M4VP6FXTEWGC6UOTY3A3V5SESS/graph.json","events_json":"https://pith.science/api/pith-number/M4VP6FXTEWGC6UOTY3A3V5SESS/events.json","paper":"https://pith.science/paper/M4VP6FXT"},"agent_actions":{"view_html":"https://pith.science/pith/M4VP6FXTEWGC6UOTY3A3V5SESS","download_json":"https://pith.science/pith/M4VP6FXTEWGC6UOTY3A3V5SESS.json","view_paper":"https://pith.science/paper/M4VP6FXT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.15341&json=true","fetch_graph":"https://pith.science/api/pith-number/M4VP6FXTEWGC6UOTY3A3V5SESS/graph.json","fetch_events":"https://pith.science/api/pith-number/M4VP6FXTEWGC6UOTY3A3V5SESS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M4VP6FXTEWGC6UOTY3A3V5SESS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M4VP6FXTEWGC6UOTY3A3V5SESS/action/storage_attestation","attest_author":"https://pith.science/pith/M4VP6FXTEWGC6UOTY3A3V5SESS/action/author_attestation","sign_citation":"https://pith.science/pith/M4VP6FXTEWGC6UOTY3A3V5SESS/action/citation_signature","submit_replication":"https://pith.science/pith/M4VP6FXTEWGC6UOTY3A3V5SESS/action/replication_record"}},"created_at":"2026-07-05T07:17:02.034285+00:00","updated_at":"2026-07-05T07:17:02.034285+00:00"}