{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:CWLYRXSCF7LHVWSWVJ2Z5ILJ5U","short_pith_number":"pith:CWLYRXSC","schema_version":"1.0","canonical_sha256":"159788de422fd67ada56aa759ea169ed0eb7b991e4040a261a1050975ba7e900","source":{"kind":"arxiv","id":"2607.04235","version":1},"attestation_state":"computed","paper":{"title":"Spinning Straw into Gold: Relabeling LLM Agent Trajectories in Hindsight for Successful Demonstrations","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Gang Wu, Jihyung Kil, Ruiyi Zhang, Ryan A. Rossi, Vlad I Morariu, Wanrong Zhu, Zichao Li, Zichao Wang","submitted_at":"2026-07-05T11:21:26Z","abstract_excerpt":"Large language model agents operate in partially observable, long-horizon settings where obtaining supervision remains a major bottleneck. We address this by utilizing a source of supervision overlooked in existing post-training methods: unintended yet successful goals embedded within agent rollouts. Specifically, we introduce Hindsight Supervised Learning (HSL), where an auxiliary LLM reviews each completed trajectory and relabels it with all of the natural-language goals the agent actually achieved. HSL then pairs the trajectory with its relabeled goals and uses these pairs for additional fi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.04235","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2026-07-05T11:21:26Z","cross_cats_sorted":[],"title_canon_sha256":"fd8d7c47f67f18017cccffb7bb1c4e2f89664cf5012462c13112955e47a8e2b9","abstract_canon_sha256":"1be5f5dfafd83dfdb0964d9ca05fe3c46e679225dde1a07ff0a5d012b9a5c96e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-07T02:19:04.750132Z","signature_b64":"/Bp6AFLlzNwKVd92bFSBCoqLZ5AVA37W5fOIeoIlsfZQtCJD/H9OzXDIrWtMWgkUaQWelt+/66g5o5ZMheK7DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"159788de422fd67ada56aa759ea169ed0eb7b991e4040a261a1050975ba7e900","last_reissued_at":"2026-07-07T02:19:04.749466Z","signature_status":"signed_v1","first_computed_at":"2026-07-07T02:19:04.749466Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Spinning Straw into Gold: Relabeling LLM Agent Trajectories in Hindsight for Successful Demonstrations","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Gang Wu, Jihyung Kil, Ruiyi Zhang, Ryan A. Rossi, Vlad I Morariu, Wanrong Zhu, Zichao Li, Zichao Wang","submitted_at":"2026-07-05T11:21:26Z","abstract_excerpt":"Large language model agents operate in partially observable, long-horizon settings where obtaining supervision remains a major bottleneck. We address this by utilizing a source of supervision overlooked in existing post-training methods: unintended yet successful goals embedded within agent rollouts. Specifically, we introduce Hindsight Supervised Learning (HSL), where an auxiliary LLM reviews each completed trajectory and relabels it with all of the natural-language goals the agent actually achieved. HSL then pairs the trajectory with its relabeled goals and uses these pairs for additional fi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.04235","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.04235/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.04235","created_at":"2026-07-07T02:19:04.749563+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.04235v1","created_at":"2026-07-07T02:19:04.749563+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.04235","created_at":"2026-07-07T02:19:04.749563+00:00"},{"alias_kind":"pith_short_12","alias_value":"CWLYRXSCF7LH","created_at":"2026-07-07T02:19:04.749563+00:00"},{"alias_kind":"pith_short_16","alias_value":"CWLYRXSCF7LHVWSW","created_at":"2026-07-07T02:19:04.749563+00:00"},{"alias_kind":"pith_short_8","alias_value":"CWLYRXSC","created_at":"2026-07-07T02:19:04.749563+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CWLYRXSCF7LHVWSWVJ2Z5ILJ5U","json":"https://pith.science/pith/CWLYRXSCF7LHVWSWVJ2Z5ILJ5U.json","graph_json":"https://pith.science/api/pith-number/CWLYRXSCF7LHVWSWVJ2Z5ILJ5U/graph.json","events_json":"https://pith.science/api/pith-number/CWLYRXSCF7LHVWSWVJ2Z5ILJ5U/events.json","paper":"https://pith.science/paper/CWLYRXSC"},"agent_actions":{"view_html":"https://pith.science/pith/CWLYRXSCF7LHVWSWVJ2Z5ILJ5U","download_json":"https://pith.science/pith/CWLYRXSCF7LHVWSWVJ2Z5ILJ5U.json","view_paper":"https://pith.science/paper/CWLYRXSC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.04235&json=true","fetch_graph":"https://pith.science/api/pith-number/CWLYRXSCF7LHVWSWVJ2Z5ILJ5U/graph.json","fetch_events":"https://pith.science/api/pith-number/CWLYRXSCF7LHVWSWVJ2Z5ILJ5U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CWLYRXSCF7LHVWSWVJ2Z5ILJ5U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CWLYRXSCF7LHVWSWVJ2Z5ILJ5U/action/storage_attestation","attest_author":"https://pith.science/pith/CWLYRXSCF7LHVWSWVJ2Z5ILJ5U/action/author_attestation","sign_citation":"https://pith.science/pith/CWLYRXSCF7LHVWSWVJ2Z5ILJ5U/action/citation_signature","submit_replication":"https://pith.science/pith/CWLYRXSCF7LHVWSWVJ2Z5ILJ5U/action/replication_record"}},"created_at":"2026-07-07T02:19:04.749563+00:00","updated_at":"2026-07-07T02:19:04.749563+00:00"}