{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ELCCDI4FL3C73EUZ7ULTWLNIYE","short_pith_number":"pith:ELCCDI4F","schema_version":"1.0","canonical_sha256":"22c421a3855ec5fd9299fd173b2da8c1285cfd5ee29dd2e830704b2a31dcfc3a","source":{"kind":"arxiv","id":"2301.12038","version":2},"attestation_state":"computed","paper":{"title":"STEERING: Stein Information Directed Exploration for Model-Based Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alec Koppel, Amrit Singh Bedi, Dinesh Manocha, Furong Huang, Mengdi Wang, Souradip Chakraborty","submitted_at":"2023-01-28T00:49:28Z","abstract_excerpt":"Directed Exploration is a crucial challenge in reinforcement learning (RL), especially when rewards are sparse. Information-directed sampling (IDS), which optimizes the information ratio, seeks to do so by augmenting regret with information gain. However, estimating information gain is computationally intractable or relies on restrictive assumptions which prohibit its use in many practical instances. In this work, we posit an alternative exploration incentive in terms of the integral probability metric (IPM) between a current estimate of the transition model and the unknown optimal, which unde"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.12038","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-01-28T00:49:28Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"b618e4bb00d6829ebcfae00e6a2d573504c60a1a994b8a76a2d427990acad30c","abstract_canon_sha256":"2b3910c83786d22d088a41ae311d4504638c4d0f9d74997bb38a9ecba5335738"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:52:02.075388Z","signature_b64":"6Sbj+BuOWdHRzMWfDeVjai8bok6tx4qXIG9binQ0H1PqiZB13p1AeeBrR6eN+DZI7DCxYR3WRIrPwGRKrlfzBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"22c421a3855ec5fd9299fd173b2da8c1285cfd5ee29dd2e830704b2a31dcfc3a","last_reissued_at":"2026-07-05T06:52:02.074865Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:52:02.074865Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"STEERING: Stein Information Directed Exploration for Model-Based Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alec Koppel, Amrit Singh Bedi, Dinesh Manocha, Furong Huang, Mengdi Wang, Souradip Chakraborty","submitted_at":"2023-01-28T00:49:28Z","abstract_excerpt":"Directed Exploration is a crucial challenge in reinforcement learning (RL), especially when rewards are sparse. Information-directed sampling (IDS), which optimizes the information ratio, seeks to do so by augmenting regret with information gain. However, estimating information gain is computationally intractable or relies on restrictive assumptions which prohibit its use in many practical instances. In this work, we posit an alternative exploration incentive in terms of the integral probability metric (IPM) between a current estimate of the transition model and the unknown optimal, which unde"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.12038","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.12038/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.12038","created_at":"2026-07-05T06:52:02.074933+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.12038v2","created_at":"2026-07-05T06:52:02.074933+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.12038","created_at":"2026-07-05T06:52:02.074933+00:00"},{"alias_kind":"pith_short_12","alias_value":"ELCCDI4FL3C7","created_at":"2026-07-05T06:52:02.074933+00:00"},{"alias_kind":"pith_short_16","alias_value":"ELCCDI4FL3C73EUZ","created_at":"2026-07-05T06:52:02.074933+00:00"},{"alias_kind":"pith_short_8","alias_value":"ELCCDI4F","created_at":"2026-07-05T06:52:02.074933+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.16734","citing_title":"Maximum Total Correlation Reinforcement Learning","ref_index":1995,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ELCCDI4FL3C73EUZ7ULTWLNIYE","json":"https://pith.science/pith/ELCCDI4FL3C73EUZ7ULTWLNIYE.json","graph_json":"https://pith.science/api/pith-number/ELCCDI4FL3C73EUZ7ULTWLNIYE/graph.json","events_json":"https://pith.science/api/pith-number/ELCCDI4FL3C73EUZ7ULTWLNIYE/events.json","paper":"https://pith.science/paper/ELCCDI4F"},"agent_actions":{"view_html":"https://pith.science/pith/ELCCDI4FL3C73EUZ7ULTWLNIYE","download_json":"https://pith.science/pith/ELCCDI4FL3C73EUZ7ULTWLNIYE.json","view_paper":"https://pith.science/paper/ELCCDI4F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.12038&json=true","fetch_graph":"https://pith.science/api/pith-number/ELCCDI4FL3C73EUZ7ULTWLNIYE/graph.json","fetch_events":"https://pith.science/api/pith-number/ELCCDI4FL3C73EUZ7ULTWLNIYE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ELCCDI4FL3C73EUZ7ULTWLNIYE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ELCCDI4FL3C73EUZ7ULTWLNIYE/action/storage_attestation","attest_author":"https://pith.science/pith/ELCCDI4FL3C73EUZ7ULTWLNIYE/action/author_attestation","sign_citation":"https://pith.science/pith/ELCCDI4FL3C73EUZ7ULTWLNIYE/action/citation_signature","submit_replication":"https://pith.science/pith/ELCCDI4FL3C73EUZ7ULTWLNIYE/action/replication_record"}},"created_at":"2026-07-05T06:52:02.074933+00:00","updated_at":"2026-07-05T06:52:02.074933+00:00"}