{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:AUFJ5RN5BQEFTB5ERAAT4EEZKA","short_pith_number":"pith:AUFJ5RN5","canonical_record":{"source":{"id":"2506.20061","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-24T23:49:28Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"5470f4187ef814e57cd55ad706d1ccb335106e57b2a8fe06729d0c0c171cf163","abstract_canon_sha256":"4d997bb8108c3749332347292ba5fe12a8781b2b219637bc409d314cc9111a42"},"schema_version":"1.0"},"canonical_sha256":"050a9ec5bd0c085987a488013e10995008985470db3dd1b5ee3a3326561ac273","source":{"kind":"arxiv","id":"2506.20061","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.20061","created_at":"2026-07-05T11:27:14Z"},{"alias_kind":"arxiv_version","alias_value":"2506.20061v1","created_at":"2026-07-05T11:27:14Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.20061","created_at":"2026-07-05T11:27:14Z"},{"alias_kind":"pith_short_12","alias_value":"AUFJ5RN5BQEF","created_at":"2026-07-05T11:27:14Z"},{"alias_kind":"pith_short_16","alias_value":"AUFJ5RN5BQEFTB5E","created_at":"2026-07-05T11:27:14Z"},{"alias_kind":"pith_short_8","alias_value":"AUFJ5RN5","created_at":"2026-07-05T11:27:14Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:AUFJ5RN5BQEFTB5ERAAT4EEZKA","target":"record","payload":{"canonical_record":{"source":{"id":"2506.20061","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-24T23:49:28Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"5470f4187ef814e57cd55ad706d1ccb335106e57b2a8fe06729d0c0c171cf163","abstract_canon_sha256":"4d997bb8108c3749332347292ba5fe12a8781b2b219637bc409d314cc9111a42"},"schema_version":"1.0"},"canonical_sha256":"050a9ec5bd0c085987a488013e10995008985470db3dd1b5ee3a3326561ac273","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:27:14.657839Z","signature_b64":"ycQnXi5K7uqyCP0zUd6OIxiNrRK16b71Tas9UYXAu6OSYJz/fABmLTezKL9gk0urPvAxnf9Cs2RRUCZyhqoSCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"050a9ec5bd0c085987a488013e10995008985470db3dd1b5ee3a3326561ac273","last_reissued_at":"2026-07-05T11:27:14.657389Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:27:14.657389Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2506.20061","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:27:14Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"YAmIwM64R58bCFZqGdltu9tbkcnxg5X3XMFGyLz7/RIL1V+7/aKxdpznVktwrVoEMgrrQaeS8PQic+BpoAGJBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T02:12:39.726316Z"},"content_sha256":"db8123c341f0de9f8e08bb1ee44eb0348f21dbceb02445a3d366bd882064af51","schema_version":"1.0","event_id":"sha256:db8123c341f0de9f8e08bb1ee44eb0348f21dbceb02445a3d366bd882064af51"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:AUFJ5RN5BQEFTB5ERAAT4EEZKA","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Learning Instruction-Following Policies through Open-Ended Instruction Relabeling with Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Fei Fang, Yali Du, Zhicheng Zhang, Ziyan Wang","submitted_at":"2025-06-24T23:49:28Z","abstract_excerpt":"Developing effective instruction-following policies in reinforcement learning remains challenging due to the reliance on extensive human-labeled instruction datasets and the difficulty of learning from sparse rewards. In this paper, we propose a novel approach that leverages the capabilities of large language models (LLMs) to automatically generate open-ended instructions retrospectively from previously collected agent trajectories. Our core idea is to employ LLMs to relabel unsuccessful trajectories by identifying meaningful subtasks the agent has implicitly accomplished, thereby enriching th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.20061","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.20061/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:27:14Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"N36Yec0FOIzS0OPCA9JsesHXzQ5kXzFzk1jUNLcBlFXAHeLiW8zRV0d1+Uw9Kl38ungA3cLPGFqY0fx/UJO4Cw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T02:12:39.726846Z"},"content_sha256":"cb49f368132f07b2873f3d5948658cd83d1b5784c95e30028dd8b47aa24e5c9e","schema_version":"1.0","event_id":"sha256:cb49f368132f07b2873f3d5948658cd83d1b5784c95e30028dd8b47aa24e5c9e"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/AUFJ5RN5BQEFTB5ERAAT4EEZKA/bundle.json","state_url":"https://pith.science/pith/AUFJ5RN5BQEFTB5ERAAT4EEZKA/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/AUFJ5RN5BQEFTB5ERAAT4EEZKA/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T02:12:39Z","links":{"resolver":"https://pith.science/pith/AUFJ5RN5BQEFTB5ERAAT4EEZKA","bundle":"https://pith.science/pith/AUFJ5RN5BQEFTB5ERAAT4EEZKA/bundle.json","state":"https://pith.science/pith/AUFJ5RN5BQEFTB5ERAAT4EEZKA/state.json","well_known_bundle":"https://pith.science/.well-known/pith/AUFJ5RN5BQEFTB5ERAAT4EEZKA/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:AUFJ5RN5BQEFTB5ERAAT4EEZKA","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"4d997bb8108c3749332347292ba5fe12a8781b2b219637bc409d314cc9111a42","cross_cats_sorted":["cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-24T23:49:28Z","title_canon_sha256":"5470f4187ef814e57cd55ad706d1ccb335106e57b2a8fe06729d0c0c171cf163"},"schema_version":"1.0","source":{"id":"2506.20061","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.20061","created_at":"2026-07-05T11:27:14Z"},{"alias_kind":"arxiv_version","alias_value":"2506.20061v1","created_at":"2026-07-05T11:27:14Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.20061","created_at":"2026-07-05T11:27:14Z"},{"alias_kind":"pith_short_12","alias_value":"AUFJ5RN5BQEF","created_at":"2026-07-05T11:27:14Z"},{"alias_kind":"pith_short_16","alias_value":"AUFJ5RN5BQEFTB5E","created_at":"2026-07-05T11:27:14Z"},{"alias_kind":"pith_short_8","alias_value":"AUFJ5RN5","created_at":"2026-07-05T11:27:14Z"}],"graph_snapshots":[{"event_id":"sha256:cb49f368132f07b2873f3d5948658cd83d1b5784c95e30028dd8b47aa24e5c9e","target":"graph","created_at":"2026-07-05T11:27:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2506.20061/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Developing effective instruction-following policies in reinforcement learning remains challenging due to the reliance on extensive human-labeled instruction datasets and the difficulty of learning from sparse rewards. In this paper, we propose a novel approach that leverages the capabilities of large language models (LLMs) to automatically generate open-ended instructions retrospectively from previously collected agent trajectories. Our core idea is to employ LLMs to relabel unsuccessful trajectories by identifying meaningful subtasks the agent has implicitly accomplished, thereby enriching th","authors_text":"Fei Fang, Yali Du, Zhicheng Zhang, Ziyan Wang","cross_cats":["cs.CL"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-24T23:49:28Z","title":"Learning Instruction-Following Policies through Open-Ended Instruction Relabeling with Large Language Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.20061","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:db8123c341f0de9f8e08bb1ee44eb0348f21dbceb02445a3d366bd882064af51","target":"record","created_at":"2026-07-05T11:27:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"4d997bb8108c3749332347292ba5fe12a8781b2b219637bc409d314cc9111a42","cross_cats_sorted":["cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-24T23:49:28Z","title_canon_sha256":"5470f4187ef814e57cd55ad706d1ccb335106e57b2a8fe06729d0c0c171cf163"},"schema_version":"1.0","source":{"id":"2506.20061","kind":"arxiv","version":1}},"canonical_sha256":"050a9ec5bd0c085987a488013e10995008985470db3dd1b5ee3a3326561ac273","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"050a9ec5bd0c085987a488013e10995008985470db3dd1b5ee3a3326561ac273","first_computed_at":"2026-07-05T11:27:14.657389Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:27:14.657389Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"ycQnXi5K7uqyCP0zUd6OIxiNrRK16b71Tas9UYXAu6OSYJz/fABmLTezKL9gk0urPvAxnf9Cs2RRUCZyhqoSCg==","signature_status":"signed_v1","signed_at":"2026-07-05T11:27:14.657839Z","signed_message":"canonical_sha256_bytes"},"source_id":"2506.20061","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:db8123c341f0de9f8e08bb1ee44eb0348f21dbceb02445a3d366bd882064af51","sha256:cb49f368132f07b2873f3d5948658cd83d1b5784c95e30028dd8b47aa24e5c9e"],"state_sha256":"021396a45352e5c6ff7b94cf016809cc7c9cd9939a081b8b3f0784c8a007f7ba"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"dqZbFRONRO4rPHsWjIbo7GTNuMmyTAASAdI8t001/cDWOfPDLWS1Kpvs2msm/5LMY1K3HJ3f7v3M6vGsqVh/Dw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T02:12:39.731988Z","bundle_sha256":"1cec61ae35368bd4ac37503efe748440d41f9c074e6b895796cd34b52db36556"}}