{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:A7GRGEQX5RFXJT5EZM2VH3H7ZZ","short_pith_number":"pith:A7GRGEQX","schema_version":"1.0","canonical_sha256":"07cd131217ec4b74cfa4cb3553ecffce61ef8e5a76b45729225357bb4798ef72","source":{"kind":"arxiv","id":"2303.06135","version":2},"attestation_state":"computed","paper":{"title":"Rewarding Chatbots for Real-World Engagement with Millions of Users","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Adian Liusie, Aliaksei Korshuk, Christie-Carol Beauchamp, Douglas Boubert, Fritz Cremer, Robert Irvine, Thomas Rialan, Valentin Assassi, Vineet Mudupalli, Vyas Raina, William Beauchamp, Xiaoding Lu, Ziyi Zhu, Zongyi Liu","submitted_at":"2023-03-10T18:53:52Z","abstract_excerpt":"The emergence of pretrained large language models has led to the deployment of a range of social chatbots for chitchat. Although these chatbots demonstrate language ability and fluency, they are not guaranteed to be engaging and can struggle to retain users. This work investigates the development of social chatbots that prioritize user engagement to enhance retention, specifically examining the use of human feedback to efficiently develop highly engaging chatbots. The proposed approach uses automatic pseudo-labels collected from user interactions to train a reward model that can be used to rej"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.06135","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-03-10T18:53:52Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"ccb76ae4e291d36e281b9979453ff3b69a8983e0f18d126164496f170083fe20","abstract_canon_sha256":"41416ce45f5eafc399a7b00b91b94082ff8f329c37257c367a4c3709960674d6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:56:41.849398Z","signature_b64":"8P9VBgKSxtIs+Nsf2HWnY3l4nvNQmwNeP9wsv+Fioap08WNTMD2U84eRKxEp/7j5gjZhfTDbzLisEEZbR3nBCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"07cd131217ec4b74cfa4cb3553ecffce61ef8e5a76b45729225357bb4798ef72","last_reissued_at":"2026-07-05T05:56:41.848875Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:56:41.848875Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Rewarding Chatbots for Real-World Engagement with Millions of Users","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Adian Liusie, Aliaksei Korshuk, Christie-Carol Beauchamp, Douglas Boubert, Fritz Cremer, Robert Irvine, Thomas Rialan, Valentin Assassi, Vineet Mudupalli, Vyas Raina, William Beauchamp, Xiaoding Lu, Ziyi Zhu, Zongyi Liu","submitted_at":"2023-03-10T18:53:52Z","abstract_excerpt":"The emergence of pretrained large language models has led to the deployment of a range of social chatbots for chitchat. Although these chatbots demonstrate language ability and fluency, they are not guaranteed to be engaging and can struggle to retain users. This work investigates the development of social chatbots that prioritize user engagement to enhance retention, specifically examining the use of human feedback to efficiently develop highly engaging chatbots. The proposed approach uses automatic pseudo-labels collected from user interactions to train a reward model that can be used to rej"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.06135","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.06135/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.06135","created_at":"2026-07-05T05:56:41.848943+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.06135v2","created_at":"2026-07-05T05:56:41.848943+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.06135","created_at":"2026-07-05T05:56:41.848943+00:00"},{"alias_kind":"pith_short_12","alias_value":"A7GRGEQX5RFX","created_at":"2026-07-05T05:56:41.848943+00:00"},{"alias_kind":"pith_short_16","alias_value":"A7GRGEQX5RFXJT5E","created_at":"2026-07-05T05:56:41.848943+00:00"},{"alias_kind":"pith_short_8","alias_value":"A7GRGEQX","created_at":"2026-07-05T05:56:41.848943+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.00084","citing_title":"Learning to Refine: Self-Refinement of Parallel Reasoning in LLMs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2309.11495","citing_title":"Chain-of-Verification Reduces Hallucination in Large Language Models","ref_index":142,"is_internal_anchor":false},{"citing_arxiv_id":"2407.21787","citing_title":"Large Language Monkeys: Scaling Inference Compute with Repeated Sampling","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17433","citing_title":"Self-Consistency from Only Two Samples: CoT-PoT Ensembling for Efficient LLM Reasoning","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A7GRGEQX5RFXJT5EZM2VH3H7ZZ","json":"https://pith.science/pith/A7GRGEQX5RFXJT5EZM2VH3H7ZZ.json","graph_json":"https://pith.science/api/pith-number/A7GRGEQX5RFXJT5EZM2VH3H7ZZ/graph.json","events_json":"https://pith.science/api/pith-number/A7GRGEQX5RFXJT5EZM2VH3H7ZZ/events.json","paper":"https://pith.science/paper/A7GRGEQX"},"agent_actions":{"view_html":"https://pith.science/pith/A7GRGEQX5RFXJT5EZM2VH3H7ZZ","download_json":"https://pith.science/pith/A7GRGEQX5RFXJT5EZM2VH3H7ZZ.json","view_paper":"https://pith.science/paper/A7GRGEQX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.06135&json=true","fetch_graph":"https://pith.science/api/pith-number/A7GRGEQX5RFXJT5EZM2VH3H7ZZ/graph.json","fetch_events":"https://pith.science/api/pith-number/A7GRGEQX5RFXJT5EZM2VH3H7ZZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A7GRGEQX5RFXJT5EZM2VH3H7ZZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A7GRGEQX5RFXJT5EZM2VH3H7ZZ/action/storage_attestation","attest_author":"https://pith.science/pith/A7GRGEQX5RFXJT5EZM2VH3H7ZZ/action/author_attestation","sign_citation":"https://pith.science/pith/A7GRGEQX5RFXJT5EZM2VH3H7ZZ/action/citation_signature","submit_replication":"https://pith.science/pith/A7GRGEQX5RFXJT5EZM2VH3H7ZZ/action/replication_record"}},"created_at":"2026-07-05T05:56:41.848943+00:00","updated_at":"2026-07-05T05:56:41.848943+00:00"}