{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:Q6O7JZUZNLAQQG4MP75MNV7PLR","short_pith_number":"pith:Q6O7JZUZ","schema_version":"1.0","canonical_sha256":"879df4e6996ac1081b8c7ffac6d7ef5c6ab2f51a9ead519c26046745f5cf40f2","source":{"kind":"arxiv","id":"1910.07207","version":2},"attestation_state":"computed","paper":{"title":"Soft Actor-Critic for Discrete Action Settings","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Petros Christodoulou","submitted_at":"2019-10-16T08:11:08Z","abstract_excerpt":"Soft Actor-Critic is a state-of-the-art reinforcement learning algorithm for continuous action settings that is not applicable to discrete action settings. Many important settings involve discrete actions, however, and so here we derive an alternative version of the Soft Actor-Critic algorithm that is applicable to discrete action settings. We then show that, even without any hyperparameter tuning, it is competitive with the tuned model-free state-of-the-art on a selection of games from the Atari suite."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1910.07207","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-10-16T08:11:08Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"2735e30b1df01b87e2a8aaed5288373d7c2a5bbeefa6bb4d257b23a9548ec868","abstract_canon_sha256":"ba6d7c04e3dbdf27a69ddd435ee39b58d638de7bf67a8243295d200d2dd4d470"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:13:04.937639Z","signature_b64":"DwuU9TC2D2tMFPFBBVJeK5U10nNMZvk9Ya19Q5kx7VWp21xfcsY9Fj3X/DZ+fyZaT4n23TF0gsXcsW9Bi70iBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"879df4e6996ac1081b8c7ffac6d7ef5c6ab2f51a9ead519c26046745f5cf40f2","last_reissued_at":"2026-07-05T00:13:04.937247Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:13:04.937247Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Soft Actor-Critic for Discrete Action Settings","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Petros Christodoulou","submitted_at":"2019-10-16T08:11:08Z","abstract_excerpt":"Soft Actor-Critic is a state-of-the-art reinforcement learning algorithm for continuous action settings that is not applicable to discrete action settings. Many important settings involve discrete actions, however, and so here we derive an alternative version of the Soft Actor-Critic algorithm that is applicable to discrete action settings. We then show that, even without any hyperparameter tuning, it is competitive with the tuned model-free state-of-the-art on a selection of games from the Atari suite."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1910.07207","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1910.07207/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1910.07207","created_at":"2026-07-05T00:13:04.937302+00:00"},{"alias_kind":"arxiv_version","alias_value":"1910.07207v2","created_at":"2026-07-05T00:13:04.937302+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1910.07207","created_at":"2026-07-05T00:13:04.937302+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q6O7JZUZNLAQ","created_at":"2026-07-05T00:13:04.937302+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q6O7JZUZNLAQQG4M","created_at":"2026-07-05T00:13:04.937302+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q6O7JZUZ","created_at":"2026-07-05T00:13:04.937302+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25394","citing_title":"FactorLibrary: From Polynomials to Circuits via Recursive Subgoals","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10705","citing_title":"Event-Driven Reinforcement Learning Enables Long-Horizon Control in Semiconductor Fabrication","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09258","citing_title":"Back to the Familiar Future: Failure Recovery for VLA Policies via Pre-Imagined Milestone Selection","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05888","citing_title":"Retry Policy Gradients in Continuous Action Spaces","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06272","citing_title":"Your GFlowNet Secretly Learns an Optimal Transport Plan","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22454","citing_title":"Don't Forget the Critic: Value-Based Data Rehearsal for Multi-Cyclic Continual Reinforcement Learning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2507.11019","citing_title":"Relative Entropy Pathwise Policy Optimization","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2509.09838","citing_title":"Dissecting Discrete Soft Actor-Critic: Limitations and Principled Alternatives","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2511.04320","citing_title":"MacroNav: Multi-Task Context Representation Learning Enables Efficient Navigation in Unknown Environments","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2511.17367","citing_title":"R2PS: Worst-Case Robust Real-Time Pursuit Strategies under Partial Observability","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q6O7JZUZNLAQQG4MP75MNV7PLR","json":"https://pith.science/pith/Q6O7JZUZNLAQQG4MP75MNV7PLR.json","graph_json":"https://pith.science/api/pith-number/Q6O7JZUZNLAQQG4MP75MNV7PLR/graph.json","events_json":"https://pith.science/api/pith-number/Q6O7JZUZNLAQQG4MP75MNV7PLR/events.json","paper":"https://pith.science/paper/Q6O7JZUZ"},"agent_actions":{"view_html":"https://pith.science/pith/Q6O7JZUZNLAQQG4MP75MNV7PLR","download_json":"https://pith.science/pith/Q6O7JZUZNLAQQG4MP75MNV7PLR.json","view_paper":"https://pith.science/paper/Q6O7JZUZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1910.07207&json=true","fetch_graph":"https://pith.science/api/pith-number/Q6O7JZUZNLAQQG4MP75MNV7PLR/graph.json","fetch_events":"https://pith.science/api/pith-number/Q6O7JZUZNLAQQG4MP75MNV7PLR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q6O7JZUZNLAQQG4MP75MNV7PLR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q6O7JZUZNLAQQG4MP75MNV7PLR/action/storage_attestation","attest_author":"https://pith.science/pith/Q6O7JZUZNLAQQG4MP75MNV7PLR/action/author_attestation","sign_citation":"https://pith.science/pith/Q6O7JZUZNLAQQG4MP75MNV7PLR/action/citation_signature","submit_replication":"https://pith.science/pith/Q6O7JZUZNLAQQG4MP75MNV7PLR/action/replication_record"}},"created_at":"2026-07-05T00:13:04.937302+00:00","updated_at":"2026-07-05T00:13:04.937302+00:00"}