{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QYYAZYPGRAABPANMYXIQXDTD7D","short_pith_number":"pith:QYYAZYPG","schema_version":"1.0","canonical_sha256":"86300ce1e688001781acc5d10b8e63f8f4a7c4a636e3ace46db27733792bae89","source":{"kind":"arxiv","id":"2306.11246","version":3},"attestation_state":"computed","paper":{"title":"Deep Reinforcement Learning for Inventory Networks: Toward Reliable Policy Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Daniel Russo, Matias Alvo, Minuk Lee, Yash Kanoria","submitted_at":"2023-06-20T02:58:25Z","abstract_excerpt":"We argue that inventory management presents unique opportunities for the reliable application of deep reinforcement learning (DRL). To enable this, we emphasize and test two complementary techniques. The first is Hindsight Differentiable Policy Optimization (HDPO), which uses pathwise gradients from offline counterfactual simulations to directly and efficiently optimize policy performance. Unlike standard policy gradient methods that rely on high-variance score-function estimators, HDPO computes gradients by differentiating through the known system dynamics. Via extensive benchmarking, we show"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.11246","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-06-20T02:58:25Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"c16a6fa2b0690bdb5dd7ee0c3a860b48730cc296f2cd7d23df35b4756783de8e","abstract_canon_sha256":"6e4f33e002d18b0770bf71fd57da3f15226bf8057bf29abfc8dffa1256dc4617"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:09:08.776722Z","signature_b64":"zP4/C3+Hz+OAbAsEzbo+wtAPVZl583JIgJD0Mf49zi+w6tMggNeXXIEcYZsE8WnlaHQ005kwyU+tB3DBVe+8AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"86300ce1e688001781acc5d10b8e63f8f4a7c4a636e3ace46db27733792bae89","last_reissued_at":"2026-07-05T12:09:08.776215Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:09:08.776215Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deep Reinforcement Learning for Inventory Networks: Toward Reliable Policy Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Daniel Russo, Matias Alvo, Minuk Lee, Yash Kanoria","submitted_at":"2023-06-20T02:58:25Z","abstract_excerpt":"We argue that inventory management presents unique opportunities for the reliable application of deep reinforcement learning (DRL). To enable this, we emphasize and test two complementary techniques. The first is Hindsight Differentiable Policy Optimization (HDPO), which uses pathwise gradients from offline counterfactual simulations to directly and efficiently optimize policy performance. Unlike standard policy gradient methods that rely on high-variance score-function estimators, HDPO computes gradients by differentiating through the known system dynamics. Via extensive benchmarking, we show"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.11246","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.11246/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.11246","created_at":"2026-07-05T12:09:08.776270+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.11246v3","created_at":"2026-07-05T12:09:08.776270+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.11246","created_at":"2026-07-05T12:09:08.776270+00:00"},{"alias_kind":"pith_short_12","alias_value":"QYYAZYPGRAAB","created_at":"2026-07-05T12:09:08.776270+00:00"},{"alias_kind":"pith_short_16","alias_value":"QYYAZYPGRAABPANM","created_at":"2026-07-05T12:09:08.776270+00:00"},{"alias_kind":"pith_short_8","alias_value":"QYYAZYPG","created_at":"2026-07-05T12:09:08.776270+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.13900","citing_title":"Ready from Day 1: Population-Aware Coordination for Large-Scale Constrained Multi-Agent Systems","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13900","citing_title":"Ready from Day 1: Population-Aware Coordination for Large-Scale Constrained Multi-Agent Systems","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14297","citing_title":"Policy Optimization in Hybrid Discrete-Continuous Action Spaces via Mixed Gradients","ref_index":119,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QYYAZYPGRAABPANMYXIQXDTD7D","json":"https://pith.science/pith/QYYAZYPGRAABPANMYXIQXDTD7D.json","graph_json":"https://pith.science/api/pith-number/QYYAZYPGRAABPANMYXIQXDTD7D/graph.json","events_json":"https://pith.science/api/pith-number/QYYAZYPGRAABPANMYXIQXDTD7D/events.json","paper":"https://pith.science/paper/QYYAZYPG"},"agent_actions":{"view_html":"https://pith.science/pith/QYYAZYPGRAABPANMYXIQXDTD7D","download_json":"https://pith.science/pith/QYYAZYPGRAABPANMYXIQXDTD7D.json","view_paper":"https://pith.science/paper/QYYAZYPG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.11246&json=true","fetch_graph":"https://pith.science/api/pith-number/QYYAZYPGRAABPANMYXIQXDTD7D/graph.json","fetch_events":"https://pith.science/api/pith-number/QYYAZYPGRAABPANMYXIQXDTD7D/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QYYAZYPGRAABPANMYXIQXDTD7D/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QYYAZYPGRAABPANMYXIQXDTD7D/action/storage_attestation","attest_author":"https://pith.science/pith/QYYAZYPGRAABPANMYXIQXDTD7D/action/author_attestation","sign_citation":"https://pith.science/pith/QYYAZYPGRAABPANMYXIQXDTD7D/action/citation_signature","submit_replication":"https://pith.science/pith/QYYAZYPGRAABPANMYXIQXDTD7D/action/replication_record"}},"created_at":"2026-07-05T12:09:08.776270+00:00","updated_at":"2026-07-05T12:09:08.776270+00:00"}