{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:JWWLFYNH5FIGSV42PPH7ROKURR","short_pith_number":"pith:JWWLFYNH","schema_version":"1.0","canonical_sha256":"4dacb2e1a7e95069579a7bcff8b9548c7e7d8cb3c309d86d7f430868f6d8b3f0","source":{"kind":"arxiv","id":"2207.11762","version":2},"attestation_state":"computed","paper":{"title":"Anti-Overestimation Dialogue Policy Learning for Task-Completion Dialogue System","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chang Tian, Marie-Francine Moens, Wenpeng Yin","submitted_at":"2022-07-24T15:38:08Z","abstract_excerpt":"A dialogue policy module is an essential part of task-completion dialogue systems. Recently, increasing interest has focused on reinforcement learning (RL)-based dialogue policy. Its favorable performance and wise action decisions rely on an accurate estimation of action values. The overestimation problem is a widely known issue of RL since its estimate of the maximum action value is larger than the ground truth, which results in an unstable learning process and suboptimal policy. This problem is detrimental to RL-based dialogue policy learning. To mitigate this problem, this paper proposes a "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2207.11762","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-07-24T15:38:08Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a1f3b565fc3d0d8bc47a3fc347173940978a3dbda6e7dc75630aa902f85b7c03","abstract_canon_sha256":"38e0645a8e8edd0c3cdf8c6b27af82c38c81f452f56776d6df0d0e7904be8291"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:07:25.968842Z","signature_b64":"RitN3V6ZIuDp2NuBI0btVTxSobA6HT28ZLO4xjYOZrAoJWTQIgfcL6+/hdFg8szb/aGTi/njYqWwHHcivtAoDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4dacb2e1a7e95069579a7bcff8b9548c7e7d8cb3c309d86d7f430868f6d8b3f0","last_reissued_at":"2026-07-05T08:07:25.968340Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:07:25.968340Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Anti-Overestimation Dialogue Policy Learning for Task-Completion Dialogue System","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chang Tian, Marie-Francine Moens, Wenpeng Yin","submitted_at":"2022-07-24T15:38:08Z","abstract_excerpt":"A dialogue policy module is an essential part of task-completion dialogue systems. Recently, increasing interest has focused on reinforcement learning (RL)-based dialogue policy. Its favorable performance and wise action decisions rely on an accurate estimation of action values. The overestimation problem is a widely known issue of RL since its estimate of the maximum action value is larger than the ground truth, which results in an unstable learning process and suboptimal policy. This problem is detrimental to RL-based dialogue policy learning. To mitigate this problem, this paper proposes a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2207.11762","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2207.11762/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2207.11762","created_at":"2026-07-05T08:07:25.968402+00:00"},{"alias_kind":"arxiv_version","alias_value":"2207.11762v2","created_at":"2026-07-05T08:07:25.968402+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2207.11762","created_at":"2026-07-05T08:07:25.968402+00:00"},{"alias_kind":"pith_short_12","alias_value":"JWWLFYNH5FIG","created_at":"2026-07-05T08:07:25.968402+00:00"},{"alias_kind":"pith_short_16","alias_value":"JWWLFYNH5FIGSV42","created_at":"2026-07-05T08:07:25.968402+00:00"},{"alias_kind":"pith_short_8","alias_value":"JWWLFYNH","created_at":"2026-07-05T08:07:25.968402+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.04848","citing_title":"Large Language Models Reasoning Abilities Under Non-Ideal Conditions After RL-Fine-Tuning","ref_index":2022,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JWWLFYNH5FIGSV42PPH7ROKURR","json":"https://pith.science/pith/JWWLFYNH5FIGSV42PPH7ROKURR.json","graph_json":"https://pith.science/api/pith-number/JWWLFYNH5FIGSV42PPH7ROKURR/graph.json","events_json":"https://pith.science/api/pith-number/JWWLFYNH5FIGSV42PPH7ROKURR/events.json","paper":"https://pith.science/paper/JWWLFYNH"},"agent_actions":{"view_html":"https://pith.science/pith/JWWLFYNH5FIGSV42PPH7ROKURR","download_json":"https://pith.science/pith/JWWLFYNH5FIGSV42PPH7ROKURR.json","view_paper":"https://pith.science/paper/JWWLFYNH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2207.11762&json=true","fetch_graph":"https://pith.science/api/pith-number/JWWLFYNH5FIGSV42PPH7ROKURR/graph.json","fetch_events":"https://pith.science/api/pith-number/JWWLFYNH5FIGSV42PPH7ROKURR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JWWLFYNH5FIGSV42PPH7ROKURR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JWWLFYNH5FIGSV42PPH7ROKURR/action/storage_attestation","attest_author":"https://pith.science/pith/JWWLFYNH5FIGSV42PPH7ROKURR/action/author_attestation","sign_citation":"https://pith.science/pith/JWWLFYNH5FIGSV42PPH7ROKURR/action/citation_signature","submit_replication":"https://pith.science/pith/JWWLFYNH5FIGSV42PPH7ROKURR/action/replication_record"}},"created_at":"2026-07-05T08:07:25.968402+00:00","updated_at":"2026-07-05T08:07:25.968402+00:00"}