{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:7OOG4IA3HNHDYYPLFW55NR3JNA","short_pith_number":"pith:7OOG4IA3","schema_version":"1.0","canonical_sha256":"fb9c6e201b3b4e3c61eb2dbbd6c769682c26f9a2370fb6954bc8431bb14a62ea","source":{"kind":"arxiv","id":"2311.05584","version":1},"attestation_state":"computed","paper":{"title":"Zero-Shot Goal-Directed Dialogue via RL on Imagined Conversations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Anca Dragan, Joey Hong, Sergey Levine","submitted_at":"2023-11-09T18:45:16Z","abstract_excerpt":"Large language models (LLMs) have emerged as powerful and general solutions to many natural language tasks. However, many of the most important applications of language generation are interactive, where an agent has to talk to a person to reach a desired outcome. For example, a teacher might try to understand their student's current comprehension level to tailor their instruction accordingly, and a travel agent might ask questions of their customer to understand their preferences in order to recommend activities they might enjoy. LLMs trained with supervised fine-tuning or \"single-step\" RL, as"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.05584","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-11-09T18:45:16Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"90ca11ee2eb9e962e80586dc71bcbbcb0f87253b5732df4ff101d44f3d54ab41","abstract_canon_sha256":"7d448a20f19fe441a05b5d91184090afd8b6d0f4f3d0cee4b1fe3349b6a32ac4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:11:09.708336Z","signature_b64":"HJf04jD7mLwVJY8ufcB9XtgWhvxXtdV0FVIaXXZPa3opb+BfU3joddEBh0cBI8QurZZTaE/89kCfMh8y6oaaCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fb9c6e201b3b4e3c61eb2dbbd6c769682c26f9a2370fb6954bc8431bb14a62ea","last_reissued_at":"2026-07-05T07:11:09.707874Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:11:09.707874Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Zero-Shot Goal-Directed Dialogue via RL on Imagined Conversations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Anca Dragan, Joey Hong, Sergey Levine","submitted_at":"2023-11-09T18:45:16Z","abstract_excerpt":"Large language models (LLMs) have emerged as powerful and general solutions to many natural language tasks. However, many of the most important applications of language generation are interactive, where an agent has to talk to a person to reach a desired outcome. For example, a teacher might try to understand their student's current comprehension level to tailor their instruction accordingly, and a travel agent might ask questions of their customer to understand their preferences in order to recommend activities they might enjoy. LLMs trained with supervised fine-tuning or \"single-step\" RL, as"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.05584","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.05584/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.05584","created_at":"2026-07-05T07:11:09.707936+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.05584v1","created_at":"2026-07-05T07:11:09.707936+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.05584","created_at":"2026-07-05T07:11:09.707936+00:00"},{"alias_kind":"pith_short_12","alias_value":"7OOG4IA3HNHD","created_at":"2026-07-05T07:11:09.707936+00:00"},{"alias_kind":"pith_short_16","alias_value":"7OOG4IA3HNHDYYPL","created_at":"2026-07-05T07:11:09.707936+00:00"},{"alias_kind":"pith_short_8","alias_value":"7OOG4IA3","created_at":"2026-07-05T07:11:09.707936+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04273","citing_title":"Characterizing initial human-AI proof formalization workflows","ref_index":72,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7OOG4IA3HNHDYYPLFW55NR3JNA","json":"https://pith.science/pith/7OOG4IA3HNHDYYPLFW55NR3JNA.json","graph_json":"https://pith.science/api/pith-number/7OOG4IA3HNHDYYPLFW55NR3JNA/graph.json","events_json":"https://pith.science/api/pith-number/7OOG4IA3HNHDYYPLFW55NR3JNA/events.json","paper":"https://pith.science/paper/7OOG4IA3"},"agent_actions":{"view_html":"https://pith.science/pith/7OOG4IA3HNHDYYPLFW55NR3JNA","download_json":"https://pith.science/pith/7OOG4IA3HNHDYYPLFW55NR3JNA.json","view_paper":"https://pith.science/paper/7OOG4IA3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.05584&json=true","fetch_graph":"https://pith.science/api/pith-number/7OOG4IA3HNHDYYPLFW55NR3JNA/graph.json","fetch_events":"https://pith.science/api/pith-number/7OOG4IA3HNHDYYPLFW55NR3JNA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7OOG4IA3HNHDYYPLFW55NR3JNA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7OOG4IA3HNHDYYPLFW55NR3JNA/action/storage_attestation","attest_author":"https://pith.science/pith/7OOG4IA3HNHDYYPLFW55NR3JNA/action/author_attestation","sign_citation":"https://pith.science/pith/7OOG4IA3HNHDYYPLFW55NR3JNA/action/citation_signature","submit_replication":"https://pith.science/pith/7OOG4IA3HNHDYYPLFW55NR3JNA/action/replication_record"}},"created_at":"2026-07-05T07:11:09.707936+00:00","updated_at":"2026-07-05T07:11:09.707936+00:00"}