{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:7N3P4VFZWMXAIBLFCBQEKEP2QK","short_pith_number":"pith:7N3P4VFZ","schema_version":"1.0","canonical_sha256":"fb76fe54b9b32e04056510604511fa828f9ca1e2ec832de568d65c4b0778026f","source":{"kind":"arxiv","id":"2210.03137","version":3},"attestation_state":"computed","paper":{"title":"Deep Inventory Management","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC"],"primary_cat":"cs.LG","authors_text":"Anna Luo, Carson Eisenach, Dean P. Foster, Dhruv Madeka, Kari Torkkola, Sham M. Kakade","submitted_at":"2022-10-06T18:00:25Z","abstract_excerpt":"This work provides a Deep Reinforcement Learning approach to solving a periodic review inventory control system with stochastic vendor lead times, lost sales, correlated demand, and price matching. While this dynamic program has historically been considered intractable, our results show that several policy learning approaches are competitive with or outperform classical methods. In order to train these algorithms, we develop novel techniques to convert historical data into a simulator. On the theoretical side, we present learnability results on a subclass of inventory control problems, where w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.03137","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-10-06T18:00:25Z","cross_cats_sorted":["math.OC"],"title_canon_sha256":"efb640d825fe356dc2b58f888e350d7094cd2371406d7391d0656e33f1a4eac5","abstract_canon_sha256":"591febe9c0ab573a530239b37d800f3cc281aa5bc9fdc0b0bb8e698a802db10b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:20:14.738941Z","signature_b64":"gxg88HWsv6cxsX0zxGfT3v66TgfdU37uO2mHpu1gkVeiishNJzkpngx1/HgwKL+1/N5j5IGZvaaiemsSk/tCBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fb76fe54b9b32e04056510604511fa828f9ca1e2ec832de568d65c4b0778026f","last_reissued_at":"2026-07-05T05:20:14.738446Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:20:14.738446Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deep Inventory Management","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC"],"primary_cat":"cs.LG","authors_text":"Anna Luo, Carson Eisenach, Dean P. Foster, Dhruv Madeka, Kari Torkkola, Sham M. Kakade","submitted_at":"2022-10-06T18:00:25Z","abstract_excerpt":"This work provides a Deep Reinforcement Learning approach to solving a periodic review inventory control system with stochastic vendor lead times, lost sales, correlated demand, and price matching. While this dynamic program has historically been considered intractable, our results show that several policy learning approaches are competitive with or outperform classical methods. In order to train these algorithms, we develop novel techniques to convert historical data into a simulator. On the theoretical side, we present learnability results on a subclass of inventory control problems, where w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.03137","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.03137/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.03137","created_at":"2026-07-05T05:20:14.738506+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.03137v3","created_at":"2026-07-05T05:20:14.738506+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.03137","created_at":"2026-07-05T05:20:14.738506+00:00"},{"alias_kind":"pith_short_12","alias_value":"7N3P4VFZWMXA","created_at":"2026-07-05T05:20:14.738506+00:00"},{"alias_kind":"pith_short_16","alias_value":"7N3P4VFZWMXAIBLF","created_at":"2026-07-05T05:20:14.738506+00:00"},{"alias_kind":"pith_short_8","alias_value":"7N3P4VFZ","created_at":"2026-07-05T05:20:14.738506+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.13900","citing_title":"Ready from Day 1: Population-Aware Coordination for Large-Scale Constrained Multi-Agent Systems","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13900","citing_title":"Ready from Day 1: Population-Aware Coordination for Large-Scale Constrained Multi-Agent Systems","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14297","citing_title":"Policy Optimization in Hybrid Discrete-Continuous Action Spaces via Mixed Gradients","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7N3P4VFZWMXAIBLFCBQEKEP2QK","json":"https://pith.science/pith/7N3P4VFZWMXAIBLFCBQEKEP2QK.json","graph_json":"https://pith.science/api/pith-number/7N3P4VFZWMXAIBLFCBQEKEP2QK/graph.json","events_json":"https://pith.science/api/pith-number/7N3P4VFZWMXAIBLFCBQEKEP2QK/events.json","paper":"https://pith.science/paper/7N3P4VFZ"},"agent_actions":{"view_html":"https://pith.science/pith/7N3P4VFZWMXAIBLFCBQEKEP2QK","download_json":"https://pith.science/pith/7N3P4VFZWMXAIBLFCBQEKEP2QK.json","view_paper":"https://pith.science/paper/7N3P4VFZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.03137&json=true","fetch_graph":"https://pith.science/api/pith-number/7N3P4VFZWMXAIBLFCBQEKEP2QK/graph.json","fetch_events":"https://pith.science/api/pith-number/7N3P4VFZWMXAIBLFCBQEKEP2QK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7N3P4VFZWMXAIBLFCBQEKEP2QK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7N3P4VFZWMXAIBLFCBQEKEP2QK/action/storage_attestation","attest_author":"https://pith.science/pith/7N3P4VFZWMXAIBLFCBQEKEP2QK/action/author_attestation","sign_citation":"https://pith.science/pith/7N3P4VFZWMXAIBLFCBQEKEP2QK/action/citation_signature","submit_replication":"https://pith.science/pith/7N3P4VFZWMXAIBLFCBQEKEP2QK/action/replication_record"}},"created_at":"2026-07-05T05:20:14.738506+00:00","updated_at":"2026-07-05T05:20:14.738506+00:00"}