{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NRCKTTR6UF557YJKE7X266FV47","short_pith_number":"pith:NRCKTTR6","schema_version":"1.0","canonical_sha256":"6c44a9ce3ea17bdfe12a27efaf78b5e7c08614256af7e35b0d4f80a6941eddf3","source":{"kind":"arxiv","id":"2504.13592","version":2},"attestation_state":"computed","paper":{"title":"Improving Generalization in Intent Detection: GRPO with Reward-Based Curriculum Sampling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baoxun Wang, Bowen Wu, Donghang Su, Qun Yu, Xiaoxue Wang, Zihao Feng, Ziwei Bai","submitted_at":"2025-04-18T09:52:12Z","abstract_excerpt":"Intent detection, a critical component in task-oriented dialogue (TOD) systems, faces significant challenges in adapting to the rapid influx of integrable tools with complex interrelationships. Existing approaches, such as zero-shot reformulations and LLM-based dynamic recognition, struggle with performance degradation when encountering unseen intents, leading to erroneous task routing. To enhance the model's generalization performance on unseen tasks, we employ Reinforcement Learning (RL) combined with a Reward-based Curriculum Sampling (RCS) during Group Relative Policy Optimization (GRPO) t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.13592","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-18T09:52:12Z","cross_cats_sorted":[],"title_canon_sha256":"1491fc03a6ab3638f4ec7bfeffaaed083c9a276fa70b9af2f42dd9ac6ebf35d1","abstract_canon_sha256":"e18010dc3c1bacfb9395332751364b8805759c0764e9d0aff8aae74328d466e9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:51:39.111335Z","signature_b64":"D4n9zo/TK4asqkghBDpRv2ysz7G9FWMKTo3YtWCg6YwYlXzPwiCdrzXB8oJzbdc3g7K5d5XiE02AJg8ejydQDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6c44a9ce3ea17bdfe12a27efaf78b5e7c08614256af7e35b0d4f80a6941eddf3","last_reissued_at":"2026-07-05T10:51:39.110773Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:51:39.110773Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving Generalization in Intent Detection: GRPO with Reward-Based Curriculum Sampling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baoxun Wang, Bowen Wu, Donghang Su, Qun Yu, Xiaoxue Wang, Zihao Feng, Ziwei Bai","submitted_at":"2025-04-18T09:52:12Z","abstract_excerpt":"Intent detection, a critical component in task-oriented dialogue (TOD) systems, faces significant challenges in adapting to the rapid influx of integrable tools with complex interrelationships. Existing approaches, such as zero-shot reformulations and LLM-based dynamic recognition, struggle with performance degradation when encountering unseen intents, leading to erroneous task routing. To enhance the model's generalization performance on unseen tasks, we employ Reinforcement Learning (RL) combined with a Reward-based Curriculum Sampling (RCS) during Group Relative Policy Optimization (GRPO) t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.13592","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.13592/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.13592","created_at":"2026-07-05T10:51:39.110834+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.13592v2","created_at":"2026-07-05T10:51:39.110834+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.13592","created_at":"2026-07-05T10:51:39.110834+00:00"},{"alias_kind":"pith_short_12","alias_value":"NRCKTTR6UF55","created_at":"2026-07-05T10:51:39.110834+00:00"},{"alias_kind":"pith_short_16","alias_value":"NRCKTTR6UF557YJK","created_at":"2026-07-05T10:51:39.110834+00:00"},{"alias_kind":"pith_short_8","alias_value":"NRCKTTR6","created_at":"2026-07-05T10:51:39.110834+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.17928","citing_title":"HEALing Entropy Collapse: Enhancing Exploration in Few-Shot RLVR via Hybrid-Domain Entropy Dynamics Alignment","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NRCKTTR6UF557YJKE7X266FV47","json":"https://pith.science/pith/NRCKTTR6UF557YJKE7X266FV47.json","graph_json":"https://pith.science/api/pith-number/NRCKTTR6UF557YJKE7X266FV47/graph.json","events_json":"https://pith.science/api/pith-number/NRCKTTR6UF557YJKE7X266FV47/events.json","paper":"https://pith.science/paper/NRCKTTR6"},"agent_actions":{"view_html":"https://pith.science/pith/NRCKTTR6UF557YJKE7X266FV47","download_json":"https://pith.science/pith/NRCKTTR6UF557YJKE7X266FV47.json","view_paper":"https://pith.science/paper/NRCKTTR6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.13592&json=true","fetch_graph":"https://pith.science/api/pith-number/NRCKTTR6UF557YJKE7X266FV47/graph.json","fetch_events":"https://pith.science/api/pith-number/NRCKTTR6UF557YJKE7X266FV47/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NRCKTTR6UF557YJKE7X266FV47/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NRCKTTR6UF557YJKE7X266FV47/action/storage_attestation","attest_author":"https://pith.science/pith/NRCKTTR6UF557YJKE7X266FV47/action/author_attestation","sign_citation":"https://pith.science/pith/NRCKTTR6UF557YJKE7X266FV47/action/citation_signature","submit_replication":"https://pith.science/pith/NRCKTTR6UF557YJKE7X266FV47/action/replication_record"}},"created_at":"2026-07-05T10:51:39.110834+00:00","updated_at":"2026-07-05T10:51:39.110834+00:00"}