{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OIKKWSVT6LZG4EGK35NJY5ZISU","short_pith_number":"pith:OIKKWSVT","schema_version":"1.0","canonical_sha256":"7214ab4ab3f2f26e10cadf5a9c7728951eac3cb2ffc71205ad4542bde23bbc4c","source":{"kind":"arxiv","id":"2411.05194","version":1},"attestation_state":"computed","paper":{"title":"Interactive Dialogue Agents via Reinforcement Learning on Hindsight Regenerations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Anca Dragan, Jessica Lin, Joey Hong, Sergey Levine","submitted_at":"2024-11-07T21:37:51Z","abstract_excerpt":"Recent progress on large language models (LLMs) has enabled dialogue agents to generate highly naturalistic and plausible text. However, current LLM language generation focuses on responding accurately to questions and requests with a single effective response. In reality, many real dialogues are interactive, meaning an agent's utterances will influence their conversational partner, elicit information, or change their opinion. Accounting for how an agent can effectively steer a conversation is a crucial ability in many dialogue tasks, from healthcare to preference elicitation. Existing methods"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.05194","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-11-07T21:37:51Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"cdd214b3cb7691ae1ed6041ed1d468240465c9b370a77e35914e276357150933","abstract_canon_sha256":"e9b1fb1edf3728b1f6ed26d0a2ceb79adffec1ca45be5796ed57e06cc00960eb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:32:55.686598Z","signature_b64":"41CoK8YFrG2Bk836szZC/KVXNFGKkd7quK8v4OYQCe4RxDsoasjVcGHoVWTQielVpXYZ0Dnq1Me2+r5LtXJYDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7214ab4ab3f2f26e10cadf5a9c7728951eac3cb2ffc71205ad4542bde23bbc4c","last_reissued_at":"2026-07-05T09:32:55.686084Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:32:55.686084Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Interactive Dialogue Agents via Reinforcement Learning on Hindsight Regenerations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Anca Dragan, Jessica Lin, Joey Hong, Sergey Levine","submitted_at":"2024-11-07T21:37:51Z","abstract_excerpt":"Recent progress on large language models (LLMs) has enabled dialogue agents to generate highly naturalistic and plausible text. However, current LLM language generation focuses on responding accurately to questions and requests with a single effective response. In reality, many real dialogues are interactive, meaning an agent's utterances will influence their conversational partner, elicit information, or change their opinion. Accounting for how an agent can effectively steer a conversation is a crucial ability in many dialogue tasks, from healthcare to preference elicitation. Existing methods"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.05194","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.05194/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.05194","created_at":"2026-07-05T09:32:55.686140+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.05194v1","created_at":"2026-07-05T09:32:55.686140+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.05194","created_at":"2026-07-05T09:32:55.686140+00:00"},{"alias_kind":"pith_short_12","alias_value":"OIKKWSVT6LZG","created_at":"2026-07-05T09:32:55.686140+00:00"},{"alias_kind":"pith_short_16","alias_value":"OIKKWSVT6LZG4EGK","created_at":"2026-07-05T09:32:55.686140+00:00"},{"alias_kind":"pith_short_8","alias_value":"OIKKWSVT","created_at":"2026-07-05T09:32:55.686140+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.24977","citing_title":"A Survey on LLM-based Conversational User Simulation","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OIKKWSVT6LZG4EGK35NJY5ZISU","json":"https://pith.science/pith/OIKKWSVT6LZG4EGK35NJY5ZISU.json","graph_json":"https://pith.science/api/pith-number/OIKKWSVT6LZG4EGK35NJY5ZISU/graph.json","events_json":"https://pith.science/api/pith-number/OIKKWSVT6LZG4EGK35NJY5ZISU/events.json","paper":"https://pith.science/paper/OIKKWSVT"},"agent_actions":{"view_html":"https://pith.science/pith/OIKKWSVT6LZG4EGK35NJY5ZISU","download_json":"https://pith.science/pith/OIKKWSVT6LZG4EGK35NJY5ZISU.json","view_paper":"https://pith.science/paper/OIKKWSVT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.05194&json=true","fetch_graph":"https://pith.science/api/pith-number/OIKKWSVT6LZG4EGK35NJY5ZISU/graph.json","fetch_events":"https://pith.science/api/pith-number/OIKKWSVT6LZG4EGK35NJY5ZISU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OIKKWSVT6LZG4EGK35NJY5ZISU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OIKKWSVT6LZG4EGK35NJY5ZISU/action/storage_attestation","attest_author":"https://pith.science/pith/OIKKWSVT6LZG4EGK35NJY5ZISU/action/author_attestation","sign_citation":"https://pith.science/pith/OIKKWSVT6LZG4EGK35NJY5ZISU/action/citation_signature","submit_replication":"https://pith.science/pith/OIKKWSVT6LZG4EGK35NJY5ZISU/action/replication_record"}},"created_at":"2026-07-05T09:32:55.686140+00:00","updated_at":"2026-07-05T09:32:55.686140+00:00"}