{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:V4M32V5VIMCZA43XOHJYRSPG4R","short_pith_number":"pith:V4M32V5V","schema_version":"1.0","canonical_sha256":"af19bd57b5430590737771d388c9e6e47765a6c327c7e40a2a39985473dc1033","source":{"kind":"arxiv","id":"2608.00133","version":1},"attestation_state":"computed","paper":{"title":"Deep Reinforcement Learning: From First Principles to Reasoning Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SY"],"primary_cat":"eess.SY","authors_text":"Ghoshana Bista","submitted_at":"2026-07-31T13:57:34Z","abstract_excerpt":"Deep reinforcement learning has evolved from classical dynamic programming, temporal-difference learning, and tabular control into a broad framework for sequential decision-making under uncertainty. This book provides a structured introduction to that evolution, emphasizing not only how reinforcement learning algorithms work, but also why they were developed, which problems they address, where they fail, and how they connect to real-world systems. It combines textbook foundations, research-oriented discussion, and a systems perspective. Early chapters introduce reinforcement learning, Markov d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.00133","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.SY","submitted_at":"2026-07-31T13:57:34Z","cross_cats_sorted":["cs.SY"],"title_canon_sha256":"8079b9aace660d29988c338d5f2620c47025419dae07abd45d1a65921129e30f","abstract_canon_sha256":"ff428cea8299923a18a50d77ef40e8e44dbf829b6a515e79a00208eb5334bcf7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-04T00:33:00.816724Z","signature_b64":"ikZ2t2JiP0yfC/W0Mt7au9ARi15ZEj5UzEqsC/+00mAOZ2SRR0jeRTGHhufh3bUL17cSZ7Mfsw67/rKdDr1gDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"af19bd57b5430590737771d388c9e6e47765a6c327c7e40a2a39985473dc1033","last_reissued_at":"2026-08-04T00:33:00.815240Z","signature_status":"signed_v1","first_computed_at":"2026-08-04T00:33:00.815240Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deep Reinforcement Learning: From First Principles to Reasoning Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SY"],"primary_cat":"eess.SY","authors_text":"Ghoshana Bista","submitted_at":"2026-07-31T13:57:34Z","abstract_excerpt":"Deep reinforcement learning has evolved from classical dynamic programming, temporal-difference learning, and tabular control into a broad framework for sequential decision-making under uncertainty. This book provides a structured introduction to that evolution, emphasizing not only how reinforcement learning algorithms work, but also why they were developed, which problems they address, where they fail, and how they connect to real-world systems. It combines textbook foundations, research-oriented discussion, and a systems perspective. Early chapters introduce reinforcement learning, Markov d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.00133","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.00133/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.00133","created_at":"2026-08-04T00:33:00.816326+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.00133v1","created_at":"2026-08-04T00:33:00.816326+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.00133","created_at":"2026-08-04T00:33:00.816326+00:00"},{"alias_kind":"pith_short_12","alias_value":"V4M32V5VIMCZ","created_at":"2026-08-04T00:33:00.816326+00:00"},{"alias_kind":"pith_short_16","alias_value":"V4M32V5VIMCZA43X","created_at":"2026-08-04T00:33:00.816326+00:00"},{"alias_kind":"pith_short_8","alias_value":"V4M32V5V","created_at":"2026-08-04T00:33:00.816326+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V4M32V5VIMCZA43XOHJYRSPG4R","json":"https://pith.science/pith/V4M32V5VIMCZA43XOHJYRSPG4R.json","graph_json":"https://pith.science/api/pith-number/V4M32V5VIMCZA43XOHJYRSPG4R/graph.json","events_json":"https://pith.science/api/pith-number/V4M32V5VIMCZA43XOHJYRSPG4R/events.json","paper":"https://pith.science/paper/V4M32V5V"},"agent_actions":{"view_html":"https://pith.science/pith/V4M32V5VIMCZA43XOHJYRSPG4R","download_json":"https://pith.science/pith/V4M32V5VIMCZA43XOHJYRSPG4R.json","view_paper":"https://pith.science/paper/V4M32V5V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.00133&json=true","fetch_graph":"https://pith.science/api/pith-number/V4M32V5VIMCZA43XOHJYRSPG4R/graph.json","fetch_events":"https://pith.science/api/pith-number/V4M32V5VIMCZA43XOHJYRSPG4R/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V4M32V5VIMCZA43XOHJYRSPG4R/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V4M32V5VIMCZA43XOHJYRSPG4R/action/storage_attestation","attest_author":"https://pith.science/pith/V4M32V5VIMCZA43XOHJYRSPG4R/action/author_attestation","sign_citation":"https://pith.science/pith/V4M32V5VIMCZA43XOHJYRSPG4R/action/citation_signature","submit_replication":"https://pith.science/pith/V4M32V5VIMCZA43XOHJYRSPG4R/action/replication_record"}},"created_at":"2026-08-04T00:33:00.816326+00:00","updated_at":"2026-08-04T00:33:00.816326+00:00"}