{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:GOWE2Y4X5HVJV3LU6JO3BEUHJY","short_pith_number":"pith:GOWE2Y4X","canonical_record":{"source":{"id":"2506.03519","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-04T03:07:55Z","cross_cats_sorted":[],"title_canon_sha256":"ba3372f3eb11a61cde8b093c5b7314047d49af1944704edaa7bbe5d832a51595","abstract_canon_sha256":"a366793c51233059d72a5721506714173b33b3cfaba6c713ca0eab493a3bd86d"},"schema_version":"1.0"},"canonical_sha256":"33ac4d6397e9ea9aed74f25db092874e1ee19a53fd7c6b1a1fcfac9021384e37","source":{"kind":"arxiv","id":"2506.03519","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.03519","created_at":"2026-07-05T11:16:09Z"},{"alias_kind":"arxiv_version","alias_value":"2506.03519v2","created_at":"2026-07-05T11:16:09Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.03519","created_at":"2026-07-05T11:16:09Z"},{"alias_kind":"pith_short_12","alias_value":"GOWE2Y4X5HVJ","created_at":"2026-07-05T11:16:09Z"},{"alias_kind":"pith_short_16","alias_value":"GOWE2Y4X5HVJV3LU","created_at":"2026-07-05T11:16:09Z"},{"alias_kind":"pith_short_8","alias_value":"GOWE2Y4X","created_at":"2026-07-05T11:16:09Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:GOWE2Y4X5HVJV3LU6JO3BEUHJY","target":"record","payload":{"canonical_record":{"source":{"id":"2506.03519","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-04T03:07:55Z","cross_cats_sorted":[],"title_canon_sha256":"ba3372f3eb11a61cde8b093c5b7314047d49af1944704edaa7bbe5d832a51595","abstract_canon_sha256":"a366793c51233059d72a5721506714173b33b3cfaba6c713ca0eab493a3bd86d"},"schema_version":"1.0"},"canonical_sha256":"33ac4d6397e9ea9aed74f25db092874e1ee19a53fd7c6b1a1fcfac9021384e37","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:09.924994Z","signature_b64":"yLRcjd2wY1PHUOQDXrvBKOxl6oJ/UvK1AHFS1lmu+oudLba8X694sXUjvAgJeZpd/W9MMI8b4yFxAyjoVkU/Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"33ac4d6397e9ea9aed74f25db092874e1ee19a53fd7c6b1a1fcfac9021384e37","last_reissued_at":"2026-07-05T11:16:09.924407Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:09.924407Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2506.03519","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:16:09Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ueSbkoMJ/7UcGOFHlvu38G2ZkfGdxOT+wy3fHGL5PuHqMKJ1e9SsJnbroid/XOLJv7L5R0UcWlwvsM6JQFKiAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T16:55:16.836833Z"},"content_sha256":"46ee23cbb83621ba9f48f9156e2824135effaeabe41414eb182b3b7842cd0238","schema_version":"1.0","event_id":"sha256:46ee23cbb83621ba9f48f9156e2824135effaeabe41414eb182b3b7842cd0238"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:GOWE2Y4X5HVJV3LU6JO3BEUHJY","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"An Efficient Task-Oriented Dialogue Policy: Evolutionary Reinforcement Learning Injected by Elite Individuals","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ben Niu, Libo Qin, Shihan Wang, Yangyang Zhao","submitted_at":"2025-06-04T03:07:55Z","abstract_excerpt":"Deep Reinforcement Learning (DRL) is widely used in task-oriented dialogue systems to optimize dialogue policy, but it struggles to balance exploration and exploitation due to the high dimensionality of state and action spaces. This challenge often results in local optima or poor convergence. Evolutionary Algorithms (EAs) have been proven to effectively explore the solution space of neural networks by maintaining population diversity. Inspired by this, we innovatively combine the global search capabilities of EA with the local optimization of DRL to achieve a balance between exploration and ex"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.03519","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.03519/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:16:09Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Q+H8UH5bjkHDyt/wZLBoNSIgXJgcre2JxVuHwIxCpxs2FozFbrfiSXFWreyPVD+mX4QkeuXTenpb+TPZUzO9Dw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T16:55:16.837330Z"},"content_sha256":"756432c6be8805c497b8df4cb38c26cb55d04a9d2c6a9457de3f6c7484f574e8","schema_version":"1.0","event_id":"sha256:756432c6be8805c497b8df4cb38c26cb55d04a9d2c6a9457de3f6c7484f574e8"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/GOWE2Y4X5HVJV3LU6JO3BEUHJY/bundle.json","state_url":"https://pith.science/pith/GOWE2Y4X5HVJV3LU6JO3BEUHJY/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/GOWE2Y4X5HVJV3LU6JO3BEUHJY/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-10T16:55:16Z","links":{"resolver":"https://pith.science/pith/GOWE2Y4X5HVJV3LU6JO3BEUHJY","bundle":"https://pith.science/pith/GOWE2Y4X5HVJV3LU6JO3BEUHJY/bundle.json","state":"https://pith.science/pith/GOWE2Y4X5HVJV3LU6JO3BEUHJY/state.json","well_known_bundle":"https://pith.science/.well-known/pith/GOWE2Y4X5HVJV3LU6JO3BEUHJY/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:GOWE2Y4X5HVJV3LU6JO3BEUHJY","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"a366793c51233059d72a5721506714173b33b3cfaba6c713ca0eab493a3bd86d","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-04T03:07:55Z","title_canon_sha256":"ba3372f3eb11a61cde8b093c5b7314047d49af1944704edaa7bbe5d832a51595"},"schema_version":"1.0","source":{"id":"2506.03519","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.03519","created_at":"2026-07-05T11:16:09Z"},{"alias_kind":"arxiv_version","alias_value":"2506.03519v2","created_at":"2026-07-05T11:16:09Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.03519","created_at":"2026-07-05T11:16:09Z"},{"alias_kind":"pith_short_12","alias_value":"GOWE2Y4X5HVJ","created_at":"2026-07-05T11:16:09Z"},{"alias_kind":"pith_short_16","alias_value":"GOWE2Y4X5HVJV3LU","created_at":"2026-07-05T11:16:09Z"},{"alias_kind":"pith_short_8","alias_value":"GOWE2Y4X","created_at":"2026-07-05T11:16:09Z"}],"graph_snapshots":[{"event_id":"sha256:756432c6be8805c497b8df4cb38c26cb55d04a9d2c6a9457de3f6c7484f574e8","target":"graph","created_at":"2026-07-05T11:16:09Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2506.03519/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Deep Reinforcement Learning (DRL) is widely used in task-oriented dialogue systems to optimize dialogue policy, but it struggles to balance exploration and exploitation due to the high dimensionality of state and action spaces. This challenge often results in local optima or poor convergence. Evolutionary Algorithms (EAs) have been proven to effectively explore the solution space of neural networks by maintaining population diversity. Inspired by this, we innovatively combine the global search capabilities of EA with the local optimization of DRL to achieve a balance between exploration and ex","authors_text":"Ben Niu, Libo Qin, Shihan Wang, Yangyang Zhao","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-04T03:07:55Z","title":"An Efficient Task-Oriented Dialogue Policy: Evolutionary Reinforcement Learning Injected by Elite Individuals"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.03519","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:46ee23cbb83621ba9f48f9156e2824135effaeabe41414eb182b3b7842cd0238","target":"record","created_at":"2026-07-05T11:16:09Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"a366793c51233059d72a5721506714173b33b3cfaba6c713ca0eab493a3bd86d","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-04T03:07:55Z","title_canon_sha256":"ba3372f3eb11a61cde8b093c5b7314047d49af1944704edaa7bbe5d832a51595"},"schema_version":"1.0","source":{"id":"2506.03519","kind":"arxiv","version":2}},"canonical_sha256":"33ac4d6397e9ea9aed74f25db092874e1ee19a53fd7c6b1a1fcfac9021384e37","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"33ac4d6397e9ea9aed74f25db092874e1ee19a53fd7c6b1a1fcfac9021384e37","first_computed_at":"2026-07-05T11:16:09.924407Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:16:09.924407Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"yLRcjd2wY1PHUOQDXrvBKOxl6oJ/UvK1AHFS1lmu+oudLba8X694sXUjvAgJeZpd/W9MMI8b4yFxAyjoVkU/Bg==","signature_status":"signed_v1","signed_at":"2026-07-05T11:16:09.924994Z","signed_message":"canonical_sha256_bytes"},"source_id":"2506.03519","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:46ee23cbb83621ba9f48f9156e2824135effaeabe41414eb182b3b7842cd0238","sha256:756432c6be8805c497b8df4cb38c26cb55d04a9d2c6a9457de3f6c7484f574e8"],"state_sha256":"99f3baa555f9e35abc8c6549c721345ce2b80b70dd8a16ea414d7c1ca5cadfb5"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"+jfM7yfb7KSBX32tlhnTZOl2eoprrjPlC2gGARAMXVv6aJZ5K0YQdlhlXlQNRW7UJn+3FeELE0KYubE4ecx2Dg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-10T16:55:16.841388Z","bundle_sha256":"54dbaf7e074742f69cd718aa728766854a013412c535f568561d05178992dd21"}}