{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:XTZD2GADRNPHINIEP66DKIUEMC","short_pith_number":"pith:XTZD2GAD","schema_version":"1.0","canonical_sha256":"bcf23d18038b5e7435047fbc352284609221801bdddf159a4bec538791b23059","source":{"kind":"arxiv","id":"2107.00602","version":1},"attestation_state":"computed","paper":{"title":"Importance Sampling based Exploration in Q Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"math.OC","authors_text":"Mort Webster, Vijay Kumar","submitted_at":"2021-07-01T16:48:35Z","abstract_excerpt":"Approximate Dynamic Programming (ADP) is a methodology to solve multi-stage stochastic optimization problems in multi-dimensional discrete or continuous spaces. ADP approximates the optimal value function by adaptively sampling both action and state space. It provides a tractable approach to very large problems, but can suffer from the exploration-exploitation dilemma. We propose a novel approach for selecting actions using importance sampling weighted by the value function approximation in continuous decision spaces to address this dilemma. An advantage of this approach is it balances explora"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2107.00602","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2021-07-01T16:48:35Z","cross_cats_sorted":[],"title_canon_sha256":"a0596aa7470c28bb088eb66c8be6a910f2bd0a2f1e6ef15502fc295a0728794f","abstract_canon_sha256":"e278f201737814cfe5084a5caa55b2603aa6d308006c22fd3e28ac02fe3e0b77"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:54:26.436529Z","signature_b64":"ydldPrQdC2tAqFWtSfCMtR1FkoSAba7pP+fYeV1Q8bMmmZef/NikSlH9zahOTL22gBFjZjH5TD36aeAbXB6dCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bcf23d18038b5e7435047fbc352284609221801bdddf159a4bec538791b23059","last_reissued_at":"2026-07-05T02:54:26.436165Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:54:26.436165Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Importance Sampling based Exploration in Q Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"math.OC","authors_text":"Mort Webster, Vijay Kumar","submitted_at":"2021-07-01T16:48:35Z","abstract_excerpt":"Approximate Dynamic Programming (ADP) is a methodology to solve multi-stage stochastic optimization problems in multi-dimensional discrete or continuous spaces. ADP approximates the optimal value function by adaptively sampling both action and state space. It provides a tractable approach to very large problems, but can suffer from the exploration-exploitation dilemma. We propose a novel approach for selecting actions using importance sampling weighted by the value function approximation in continuous decision spaces to address this dilemma. An advantage of this approach is it balances explora"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2107.00602","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2107.00602/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2107.00602","created_at":"2026-07-05T02:54:26.436224+00:00"},{"alias_kind":"arxiv_version","alias_value":"2107.00602v1","created_at":"2026-07-05T02:54:26.436224+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2107.00602","created_at":"2026-07-05T02:54:26.436224+00:00"},{"alias_kind":"pith_short_12","alias_value":"XTZD2GADRNPH","created_at":"2026-07-05T02:54:26.436224+00:00"},{"alias_kind":"pith_short_16","alias_value":"XTZD2GADRNPHINIE","created_at":"2026-07-05T02:54:26.436224+00:00"},{"alias_kind":"pith_short_8","alias_value":"XTZD2GAD","created_at":"2026-07-05T02:54:26.436224+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XTZD2GADRNPHINIEP66DKIUEMC","json":"https://pith.science/pith/XTZD2GADRNPHINIEP66DKIUEMC.json","graph_json":"https://pith.science/api/pith-number/XTZD2GADRNPHINIEP66DKIUEMC/graph.json","events_json":"https://pith.science/api/pith-number/XTZD2GADRNPHINIEP66DKIUEMC/events.json","paper":"https://pith.science/paper/XTZD2GAD"},"agent_actions":{"view_html":"https://pith.science/pith/XTZD2GADRNPHINIEP66DKIUEMC","download_json":"https://pith.science/pith/XTZD2GADRNPHINIEP66DKIUEMC.json","view_paper":"https://pith.science/paper/XTZD2GAD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2107.00602&json=true","fetch_graph":"https://pith.science/api/pith-number/XTZD2GADRNPHINIEP66DKIUEMC/graph.json","fetch_events":"https://pith.science/api/pith-number/XTZD2GADRNPHINIEP66DKIUEMC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XTZD2GADRNPHINIEP66DKIUEMC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XTZD2GADRNPHINIEP66DKIUEMC/action/storage_attestation","attest_author":"https://pith.science/pith/XTZD2GADRNPHINIEP66DKIUEMC/action/author_attestation","sign_citation":"https://pith.science/pith/XTZD2GADRNPHINIEP66DKIUEMC/action/citation_signature","submit_replication":"https://pith.science/pith/XTZD2GADRNPHINIEP66DKIUEMC/action/replication_record"}},"created_at":"2026-07-05T02:54:26.436224+00:00","updated_at":"2026-07-05T02:54:26.436224+00:00"}