{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:GICCTQRO3CZMGJHWVIW3SPCBDV","short_pith_number":"pith:GICCTQRO","schema_version":"1.0","canonical_sha256":"320429c22ed8b2c324f6aa2db93c411d7a37e8244a65e685b720426a6fe025c8","source":{"kind":"arxiv","id":"2112.02215","version":3},"attestation_state":"computed","paper":{"title":"Deep Policy Iteration with Integer Programming for Inventory Management","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","math.OC"],"primary_cat":"cs.LG","authors_text":"Ashish Jagmohan, Brian Quanz, Divya Singhvi, Jayant Kalagnanam, Pavithra Harsha","submitted_at":"2021-12-04T01:40:34Z","abstract_excerpt":"We present a Reinforcement Learning (RL) based framework for optimizing long-term discounted reward problems with large combinatorial action space and state dependent constraints. These characteristics are common to many operations management problems, e.g., network inventory replenishment, where managers have to deal with uncertain demand, lost sales, and capacity constraints that results in more complex feasible action spaces. Our proposed Programmable Actor Reinforcement Learning (PARL) uses a deep-policy iteration method that leverages neural networks (NNs) to approximate the value functio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.02215","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-12-04T01:40:34Z","cross_cats_sorted":["cs.AI","math.OC"],"title_canon_sha256":"afc39b45f3abdb12a3ccd9d33db70f2fd173c4cc5768ede02b96d8d2495a08eb","abstract_canon_sha256":"fab82c73bf839e62d947897f801e661d129960bc832ae448425a5ef1f1bbf46d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:58:15.064530Z","signature_b64":"U+FKwvZJJS4siXMEoQ9TNz9HTSS7FIRjs001a+Z41YIbsZKpWcCQzGNxQmRRJgyo+nAyVcTxj/mUBFI4cjamCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"320429c22ed8b2c324f6aa2db93c411d7a37e8244a65e685b720426a6fe025c8","last_reissued_at":"2026-07-05T09:58:15.064100Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:58:15.064100Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deep Policy Iteration with Integer Programming for Inventory Management","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","math.OC"],"primary_cat":"cs.LG","authors_text":"Ashish Jagmohan, Brian Quanz, Divya Singhvi, Jayant Kalagnanam, Pavithra Harsha","submitted_at":"2021-12-04T01:40:34Z","abstract_excerpt":"We present a Reinforcement Learning (RL) based framework for optimizing long-term discounted reward problems with large combinatorial action space and state dependent constraints. These characteristics are common to many operations management problems, e.g., network inventory replenishment, where managers have to deal with uncertain demand, lost sales, and capacity constraints that results in more complex feasible action spaces. Our proposed Programmable Actor Reinforcement Learning (PARL) uses a deep-policy iteration method that leverages neural networks (NNs) to approximate the value functio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.02215","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.02215/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.02215","created_at":"2026-07-05T09:58:15.064161+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.02215v3","created_at":"2026-07-05T09:58:15.064161+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.02215","created_at":"2026-07-05T09:58:15.064161+00:00"},{"alias_kind":"pith_short_12","alias_value":"GICCTQRO3CZM","created_at":"2026-07-05T09:58:15.064161+00:00"},{"alias_kind":"pith_short_16","alias_value":"GICCTQRO3CZMGJHW","created_at":"2026-07-05T09:58:15.064161+00:00"},{"alias_kind":"pith_short_8","alias_value":"GICCTQRO","created_at":"2026-07-05T09:58:15.064161+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GICCTQRO3CZMGJHWVIW3SPCBDV","json":"https://pith.science/pith/GICCTQRO3CZMGJHWVIW3SPCBDV.json","graph_json":"https://pith.science/api/pith-number/GICCTQRO3CZMGJHWVIW3SPCBDV/graph.json","events_json":"https://pith.science/api/pith-number/GICCTQRO3CZMGJHWVIW3SPCBDV/events.json","paper":"https://pith.science/paper/GICCTQRO"},"agent_actions":{"view_html":"https://pith.science/pith/GICCTQRO3CZMGJHWVIW3SPCBDV","download_json":"https://pith.science/pith/GICCTQRO3CZMGJHWVIW3SPCBDV.json","view_paper":"https://pith.science/paper/GICCTQRO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.02215&json=true","fetch_graph":"https://pith.science/api/pith-number/GICCTQRO3CZMGJHWVIW3SPCBDV/graph.json","fetch_events":"https://pith.science/api/pith-number/GICCTQRO3CZMGJHWVIW3SPCBDV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GICCTQRO3CZMGJHWVIW3SPCBDV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GICCTQRO3CZMGJHWVIW3SPCBDV/action/storage_attestation","attest_author":"https://pith.science/pith/GICCTQRO3CZMGJHWVIW3SPCBDV/action/author_attestation","sign_citation":"https://pith.science/pith/GICCTQRO3CZMGJHWVIW3SPCBDV/action/citation_signature","submit_replication":"https://pith.science/pith/GICCTQRO3CZMGJHWVIW3SPCBDV/action/replication_record"}},"created_at":"2026-07-05T09:58:15.064161+00:00","updated_at":"2026-07-05T09:58:15.064161+00:00"}