{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:4VEMU65TQVP3XXKPKUTFDJ273L","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"52d63c028e2ad21173ffa2c755d7be287656e22578e6960c0396222d92bf73d5","cross_cats_sorted":["cs.AI","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-31T16:52:47Z","title_canon_sha256":"9f6c219e7d5a134b73ed4c3e7ce2cbb274448e0d284194b3b76ab7abb15d012f"},"schema_version":"1.0","source":{"id":"2607.29617","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.29617","created_at":"2026-08-03T01:40:33Z"},{"alias_kind":"arxiv_version","alias_value":"2607.29617v1","created_at":"2026-08-03T01:40:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.29617","created_at":"2026-08-03T01:40:33Z"},{"alias_kind":"pith_short_12","alias_value":"4VEMU65TQVP3","created_at":"2026-08-03T01:40:33Z"},{"alias_kind":"pith_short_16","alias_value":"4VEMU65TQVP3XXKP","created_at":"2026-08-03T01:40:33Z"},{"alias_kind":"pith_short_8","alias_value":"4VEMU65T","created_at":"2026-08-03T01:40:33Z"}],"graph_snapshots":[{"event_id":"sha256:360be3d29857a061da5f0d662a1b4802090877840576944cb43b4adfc14cef90","target":"graph","created_at":"2026-08-03T01:40:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.29617/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Imitation learning (IL)---training an agent to replicate expert behavior from demonstrations---underpins applications from robotics to language model training. Standard approaches such as Behavior Cloning (BC) are known to suffer from compounding errors and performance plateaus, particularly when the learner cannot perfectly represent the expert's policy (as is typical, e.g., in distillation). Two interventions are widely understood empirically to improve performance: querying the expert interactively along the learner's own trajectories, and using value function estimation en route to generat","authors_text":"Antoine Moulin, Audrey Huang, Dylan J. Foster, Luca Viano, Philip Amortila, Volkan Cevher","cross_cats":["cs.AI","stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-31T16:52:47Z","title":"When Does On-Policy Interaction Help? Representational Tradeoffs in Value-Based Imitation Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.29617","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:5a88719f39d356464a1a747b7202d23af7449bf81bb9198d565240ff71c6b4f8","target":"record","created_at":"2026-08-03T01:40:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"52d63c028e2ad21173ffa2c755d7be287656e22578e6960c0396222d92bf73d5","cross_cats_sorted":["cs.AI","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-31T16:52:47Z","title_canon_sha256":"9f6c219e7d5a134b73ed4c3e7ce2cbb274448e0d284194b3b76ab7abb15d012f"},"schema_version":"1.0","source":{"id":"2607.29617","kind":"arxiv","version":1}},"canonical_sha256":"e548ca7bb3855fbbdd4f552651a75fdadf696ab6e82064d58b0646725f6afe75","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"e548ca7bb3855fbbdd4f552651a75fdadf696ab6e82064d58b0646725f6afe75","first_computed_at":"2026-08-03T01:40:33.373545Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-08-03T01:40:33.373545Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"r96xk0WAN0cR6kXmkDSVsxtBxamLJGAUQ7rj2kv8wX4Ubex0jaNhuMg7FDwb9sEXTIKOtUBH+1KpgZ4a5mvvDw==","signature_status":"signed_v1","signed_at":"2026-08-03T01:40:33.375051Z","signed_message":"canonical_sha256_bytes"},"source_id":"2607.29617","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:5a88719f39d356464a1a747b7202d23af7449bf81bb9198d565240ff71c6b4f8","sha256:360be3d29857a061da5f0d662a1b4802090877840576944cb43b4adfc14cef90"],"state_sha256":"e5e32957831b18ae36a98168c04ee67a376ec181e052159327ff77e9b6f07164"}