{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OAGN6RIFWG373TTDKX74YYM636","short_pith_number":"pith:OAGN6RIF","schema_version":"1.0","canonical_sha256":"700cdf4505b1b7fdce6355ffcc619edfbccf1b0e6efdb7204e5a071071b5c901","source":{"kind":"arxiv","id":"2504.21383","version":1},"attestation_state":"computed","paper":{"title":"FAST-Q: Fast-track Exploration with Adversarially Balanced State Representations for Counterfactual Action Estimation in Offline Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Aditya Pareek, Pulkit Agrawal, Rukma Talwadker, Tridib Mukherjee","submitted_at":"2025-04-30T07:32:40Z","abstract_excerpt":"Recent advancements in state-of-the-art (SOTA) offline reinforcement learning (RL) have primarily focused on addressing function approximation errors, which contribute to the overestimation of Q-values for out-of-distribution actions, a challenge that static datasets exacerbate. However, high stakes applications such as recommendation systems in online gaming, introduce further complexities due to player's psychology (intent) driven by gameplay experiences and the inherent volatility on the platform. These factors create highly sparse, partially overlapping state spaces across policies, furthe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.21383","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-30T07:32:40Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2d979a79f2180f910047156484f76c799baa9586f9a36d486e810a154c53c457","abstract_canon_sha256":"fd6e729a1351649a5316c3b6c16dbf28214863ef58e0a237b0344e8ab9a788f5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:56:23.094041Z","signature_b64":"l5qF3Jr/xmSNf/uK3BH1EvHI00p5fzCwkqrodBMO6V3roSRd/nid4orFmK+/zPKHEzcRqLhZIpowi9/VKjuADw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"700cdf4505b1b7fdce6355ffcc619edfbccf1b0e6efdb7204e5a071071b5c901","last_reissued_at":"2026-07-05T10:56:23.093618Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:56:23.093618Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FAST-Q: Fast-track Exploration with Adversarially Balanced State Representations for Counterfactual Action Estimation in Offline Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Aditya Pareek, Pulkit Agrawal, Rukma Talwadker, Tridib Mukherjee","submitted_at":"2025-04-30T07:32:40Z","abstract_excerpt":"Recent advancements in state-of-the-art (SOTA) offline reinforcement learning (RL) have primarily focused on addressing function approximation errors, which contribute to the overestimation of Q-values for out-of-distribution actions, a challenge that static datasets exacerbate. However, high stakes applications such as recommendation systems in online gaming, introduce further complexities due to player's psychology (intent) driven by gameplay experiences and the inherent volatility on the platform. These factors create highly sparse, partially overlapping state spaces across policies, furthe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.21383","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.21383/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.21383","created_at":"2026-07-05T10:56:23.093687+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.21383v1","created_at":"2026-07-05T10:56:23.093687+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.21383","created_at":"2026-07-05T10:56:23.093687+00:00"},{"alias_kind":"pith_short_12","alias_value":"OAGN6RIFWG37","created_at":"2026-07-05T10:56:23.093687+00:00"},{"alias_kind":"pith_short_16","alias_value":"OAGN6RIFWG373TTD","created_at":"2026-07-05T10:56:23.093687+00:00"},{"alias_kind":"pith_short_8","alias_value":"OAGN6RIF","created_at":"2026-07-05T10:56:23.093687+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OAGN6RIFWG373TTDKX74YYM636","json":"https://pith.science/pith/OAGN6RIFWG373TTDKX74YYM636.json","graph_json":"https://pith.science/api/pith-number/OAGN6RIFWG373TTDKX74YYM636/graph.json","events_json":"https://pith.science/api/pith-number/OAGN6RIFWG373TTDKX74YYM636/events.json","paper":"https://pith.science/paper/OAGN6RIF"},"agent_actions":{"view_html":"https://pith.science/pith/OAGN6RIFWG373TTDKX74YYM636","download_json":"https://pith.science/pith/OAGN6RIFWG373TTDKX74YYM636.json","view_paper":"https://pith.science/paper/OAGN6RIF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.21383&json=true","fetch_graph":"https://pith.science/api/pith-number/OAGN6RIFWG373TTDKX74YYM636/graph.json","fetch_events":"https://pith.science/api/pith-number/OAGN6RIFWG373TTDKX74YYM636/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OAGN6RIFWG373TTDKX74YYM636/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OAGN6RIFWG373TTDKX74YYM636/action/storage_attestation","attest_author":"https://pith.science/pith/OAGN6RIFWG373TTDKX74YYM636/action/author_attestation","sign_citation":"https://pith.science/pith/OAGN6RIFWG373TTDKX74YYM636/action/citation_signature","submit_replication":"https://pith.science/pith/OAGN6RIFWG373TTDKX74YYM636/action/replication_record"}},"created_at":"2026-07-05T10:56:23.093687+00:00","updated_at":"2026-07-05T10:56:23.093687+00:00"}