{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:HABA3NJKL4K4LGP7GBO3YLC26K","short_pith_number":"pith:HABA3NJK","canonical_record":{"source":{"id":"2605.09183","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-05-09T21:48:04Z","cross_cats_sorted":[],"title_canon_sha256":"42d90b5f92d643d087f94d6acd8fb122e875db947218acf18f0857f716e620db","abstract_canon_sha256":"b1996c93f3ed7408fc7fe39f93706ebab07ea22dfa37d3c34c74770a33f86d1e"},"schema_version":"1.0"},"canonical_sha256":"38020db52a5f15c599ff305dbc2c5af29e35b64c64aeb9998f49f50cb6a32422","source":{"kind":"arxiv","id":"2605.09183","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2605.09183","created_at":"2026-05-20T00:03:16Z"},{"alias_kind":"arxiv_version","alias_value":"2605.09183v2","created_at":"2026-05-20T00:03:16Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.09183","created_at":"2026-05-20T00:03:16Z"},{"alias_kind":"pith_short_12","alias_value":"HABA3NJKL4K4","created_at":"2026-05-20T00:03:16Z"},{"alias_kind":"pith_short_16","alias_value":"HABA3NJKL4K4LGP7","created_at":"2026-05-20T00:03:16Z"},{"alias_kind":"pith_short_8","alias_value":"HABA3NJK","created_at":"2026-05-20T00:03:16Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:HABA3NJKL4K4LGP7GBO3YLC26K","target":"record","payload":{"canonical_record":{"source":{"id":"2605.09183","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-05-09T21:48:04Z","cross_cats_sorted":[],"title_canon_sha256":"42d90b5f92d643d087f94d6acd8fb122e875db947218acf18f0857f716e620db","abstract_canon_sha256":"b1996c93f3ed7408fc7fe39f93706ebab07ea22dfa37d3c34c74770a33f86d1e"},"schema_version":"1.0"},"canonical_sha256":"38020db52a5f15c599ff305dbc2c5af29e35b64c64aeb9998f49f50cb6a32422","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-20T00:03:16.230505Z","signature_b64":"qEWJYEnlXESOWZGSVEiSFG0XZYoiJFDyfvk4dn6a8BbAPvuvziItxT2jrIY6sERE4x01KET64D/qG8C8O/PODw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"38020db52a5f15c599ff305dbc2c5af29e35b64c64aeb9998f49f50cb6a32422","last_reissued_at":"2026-05-20T00:03:16.229703Z","signature_status":"signed_v1","first_computed_at":"2026-05-20T00:03:16.229703Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2605.09183","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-20T00:03:16Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"wKyW9JblO1BHs8dQltYjammKfUULoaDz/y7SS4DHFhMRlZy+ZOpxZEoo7uphbQHl2uFfTUaovLIBgdCqfnvyCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T14:31:05.358168Z"},"content_sha256":"af6284f3f2d0ab141e8150fa6ffdbe73eb88aa92a0e91e9a77fee27322ea5ffd","schema_version":"1.0","event_id":"sha256:af6284f3f2d0ab141e8150fa6ffdbe73eb88aa92a0e91e9a77fee27322ea5ffd"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:HABA3NJKL4K4LGP7GBO3YLC26K","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Learning When to Stop: Selective Imitation Learning Under Arbitrary Dynamics Shift","license":"http://creativecommons.org/licenses/by/4.0/","headline":"Selective imitation learning lets agents stop acting when dynamics shift makes expert demonstrations unreliable, using a small set of validator policies.","cross_cats":[],"primary_cat":"cs.LG","authors_text":"James Wang, Jonathan Pei, Surbhi Goel","submitted_at":"2026-05-09T21:48:04Z","abstract_excerpt":"Behavior cloning provides strong imitation learning guarantees when training and test environments share the same dynamics. However, in many deployment settings the test environment's transitions differ from training, and classical offline IL offers no recourse: the learner must commit to an action at every state, even when its demonstrations are uninformative and could lead to arbitrary degradation of performance. This motivates the study of selective imitation, where the learner may choose to stop when it cannot act reliably. We introduce a model for selective imitation under arbitrary dynam"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Our algorithm, SeqRejectron, constructs a stopping rule using a small set of validator policies whose size is independent of the horizon or policy class. For deterministic policies, this yields horizon-free Õ(log|Π|/ε²) sample complexity, assuming sparse costs.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The assumption that unlabeled state trajectories from the same expert are available in the test environment, together with the sparse-cost assumption needed for the deterministic horizon-free bound; if either fails the stated guarantees do not apply.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"SeqRejectron builds a stopping rule from a small set of validator policies to achieve horizon-free sample-complexity guarantees for selective imitation learning under arbitrary train-test dynamics shifts.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Selective imitation learning lets agents stop acting when dynamics shift makes expert demonstrations unreliable, using a small set of validator policies.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"f93d49d700d9e92e54695bbc70decb0f7fbf52311dcc56e14165c6ccc2661d7f"},"source":{"id":"2605.09183","kind":"arxiv","version":2},"verdict":{"id":"c9c702d1-5720-4024-8760-85fef517ba82","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-12T03:18:58.353122Z","strongest_claim":"Our algorithm, SeqRejectron, constructs a stopping rule using a small set of validator policies whose size is independent of the horizon or policy class. For deterministic policies, this yields horizon-free Õ(log|Π|/ε²) sample complexity, assuming sparse costs.","one_line_summary":"SeqRejectron builds a stopping rule from a small set of validator policies to achieve horizon-free sample-complexity guarantees for selective imitation learning under arbitrary train-test dynamics shifts.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The assumption that unlabeled state trajectories from the same expert are available in the test environment, together with the sparse-cost assumption needed for the deterministic horizon-free bound; if either fails the stated guarantees do not apply.","pith_extraction_headline":"Selective imitation learning lets agents stop acting when dynamics shift makes expert demonstrations unreliable, using a small set of validator policies."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2605.09183/integrity.json","findings":[],"available":true,"detectors_run":[{"name":"ai_meta_artifact","ran_at":"2026-05-19T20:36:06.908836Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_title_agreement","ran_at":"2026-05-19T13:31:18.437032Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_compliance","ran_at":"2026-05-19T10:29:03.382813Z","status":"completed","version":"1.0.0","findings_count":0}],"snapshot_sha256":"d07ca53bacaba8279e5123ff21d5ce69fa813dea8b9125b8add6ea8898191564"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"c9c702d1-5720-4024-8760-85fef517ba82"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-20T00:03:16Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"AgYewdgDUDH+uv7wLSBJ36CeF2kyjFXBf84NIkDokuwT45WBHWiOn75tmToZc8kWTJH/Reke2gsD4HuKyu+aBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T14:31:05.358967Z"},"content_sha256":"7fe05de5ec83049aaa4102c0ed842a56ea9200469707176c9fc863f5c07ab890","schema_version":"1.0","event_id":"sha256:7fe05de5ec83049aaa4102c0ed842a56ea9200469707176c9fc863f5c07ab890"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/HABA3NJKL4K4LGP7GBO3YLC26K/bundle.json","state_url":"https://pith.science/pith/HABA3NJKL4K4LGP7GBO3YLC26K/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/HABA3NJKL4K4LGP7GBO3YLC26K/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T14:31:05Z","links":{"resolver":"https://pith.science/pith/HABA3NJKL4K4LGP7GBO3YLC26K","bundle":"https://pith.science/pith/HABA3NJKL4K4LGP7GBO3YLC26K/bundle.json","state":"https://pith.science/pith/HABA3NJKL4K4LGP7GBO3YLC26K/state.json","well_known_bundle":"https://pith.science/.well-known/pith/HABA3NJKL4K4LGP7GBO3YLC26K/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:HABA3NJKL4K4LGP7GBO3YLC26K","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"b1996c93f3ed7408fc7fe39f93706ebab07ea22dfa37d3c34c74770a33f86d1e","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-05-09T21:48:04Z","title_canon_sha256":"42d90b5f92d643d087f94d6acd8fb122e875db947218acf18f0857f716e620db"},"schema_version":"1.0","source":{"id":"2605.09183","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2605.09183","created_at":"2026-05-20T00:03:16Z"},{"alias_kind":"arxiv_version","alias_value":"2605.09183v2","created_at":"2026-05-20T00:03:16Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.09183","created_at":"2026-05-20T00:03:16Z"},{"alias_kind":"pith_short_12","alias_value":"HABA3NJKL4K4","created_at":"2026-05-20T00:03:16Z"},{"alias_kind":"pith_short_16","alias_value":"HABA3NJKL4K4LGP7","created_at":"2026-05-20T00:03:16Z"},{"alias_kind":"pith_short_8","alias_value":"HABA3NJK","created_at":"2026-05-20T00:03:16Z"}],"graph_snapshots":[{"event_id":"sha256:7fe05de5ec83049aaa4102c0ed842a56ea9200469707176c9fc863f5c07ab890","target":"graph","created_at":"2026-05-20T00:03:16Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Our algorithm, SeqRejectron, constructs a stopping rule using a small set of validator policies whose size is independent of the horizon or policy class. For deterministic policies, this yields horizon-free Õ(log|Π|/ε²) sample complexity, assuming sparse costs."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The assumption that unlabeled state trajectories from the same expert are available in the test environment, together with the sparse-cost assumption needed for the deterministic horizon-free bound; if either fails the stated guarantees do not apply."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"SeqRejectron builds a stopping rule from a small set of validator policies to achieve horizon-free sample-complexity guarantees for selective imitation learning under arbitrary train-test dynamics shifts."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Selective imitation learning lets agents stop acting when dynamics shift makes expert demonstrations unreliable, using a small set of validator policies."}],"snapshot_sha256":"f93d49d700d9e92e54695bbc70decb0f7fbf52311dcc56e14165c6ccc2661d7f"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[{"findings_count":0,"name":"ai_meta_artifact","ran_at":"2026-05-19T20:36:06.908836Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"doi_title_agreement","ran_at":"2026-05-19T13:31:18.437032Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"doi_compliance","ran_at":"2026-05-19T10:29:03.382813Z","status":"completed","version":"1.0.0"}],"endpoint":"/pith/2605.09183/integrity.json","findings":[],"snapshot_sha256":"d07ca53bacaba8279e5123ff21d5ce69fa813dea8b9125b8add6ea8898191564","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Behavior cloning provides strong imitation learning guarantees when training and test environments share the same dynamics. However, in many deployment settings the test environment's transitions differ from training, and classical offline IL offers no recourse: the learner must commit to an action at every state, even when its demonstrations are uninformative and could lead to arbitrary degradation of performance. This motivates the study of selective imitation, where the learner may choose to stop when it cannot act reliably. We introduce a model for selective imitation under arbitrary dynam","authors_text":"James Wang, Jonathan Pei, Surbhi Goel","cross_cats":[],"headline":"Selective imitation learning lets agents stop acting when dynamics shift makes expert demonstrations unreliable, using a small set of validator policies.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-05-09T21:48:04Z","title":"Learning When to Stop: Selective Imitation Learning Under Arbitrary Dynamics Shift"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2605.09183","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-12T03:18:58.353122Z","id":"c9c702d1-5720-4024-8760-85fef517ba82","model_set":{"reader":"grok-4.3"},"one_line_summary":"SeqRejectron builds a stopping rule from a small set of validator policies to achieve horizon-free sample-complexity guarantees for selective imitation learning under arbitrary train-test dynamics shifts.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Selective imitation learning lets agents stop acting when dynamics shift makes expert demonstrations unreliable, using a small set of validator policies.","strongest_claim":"Our algorithm, SeqRejectron, constructs a stopping rule using a small set of validator policies whose size is independent of the horizon or policy class. For deterministic policies, this yields horizon-free Õ(log|Π|/ε²) sample complexity, assuming sparse costs.","weakest_assumption":"The assumption that unlabeled state trajectories from the same expert are available in the test environment, together with the sparse-cost assumption needed for the deterministic horizon-free bound; if either fails the stated guarantees do not apply."}},"verdict_id":"c9c702d1-5720-4024-8760-85fef517ba82"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:af6284f3f2d0ab141e8150fa6ffdbe73eb88aa92a0e91e9a77fee27322ea5ffd","target":"record","created_at":"2026-05-20T00:03:16Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"b1996c93f3ed7408fc7fe39f93706ebab07ea22dfa37d3c34c74770a33f86d1e","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-05-09T21:48:04Z","title_canon_sha256":"42d90b5f92d643d087f94d6acd8fb122e875db947218acf18f0857f716e620db"},"schema_version":"1.0","source":{"id":"2605.09183","kind":"arxiv","version":2}},"canonical_sha256":"38020db52a5f15c599ff305dbc2c5af29e35b64c64aeb9998f49f50cb6a32422","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"38020db52a5f15c599ff305dbc2c5af29e35b64c64aeb9998f49f50cb6a32422","first_computed_at":"2026-05-20T00:03:16.229703Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-20T00:03:16.229703Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"qEWJYEnlXESOWZGSVEiSFG0XZYoiJFDyfvk4dn6a8BbAPvuvziItxT2jrIY6sERE4x01KET64D/qG8C8O/PODw==","signature_status":"signed_v1","signed_at":"2026-05-20T00:03:16.230505Z","signed_message":"canonical_sha256_bytes"},"source_id":"2605.09183","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:af6284f3f2d0ab141e8150fa6ffdbe73eb88aa92a0e91e9a77fee27322ea5ffd","sha256:7fe05de5ec83049aaa4102c0ed842a56ea9200469707176c9fc863f5c07ab890"],"state_sha256":"339fb5746f0e5feabbd05a92426d1cac93f783585876b2cfc92abbf2252c8346"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Y+9+HyAFndBGj3l7ZOs2n9k8AdZUgSKkzlQr7wkmI6jy44miWY4v7TGVWZvvsb1eiWL1IiotnmPAqoYIUYKFAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T14:31:05.364308Z","bundle_sha256":"58dfdbec428f44ee814a265a2a98f6e39198e17d5624058994eee7ec787fc5d9"}}