{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:AJR7CYNWUB5BMJ6MLI5HVHF3SN","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"951095327d0a02939cf81090632c4021aab23492e57c8f440ffb44463668f42f","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-03T23:08:42Z","title_canon_sha256":"caefdc1875359f6b5dd310e6e3b1612f58ae163913243f908e789f17172d7281"},"schema_version":"1.0","source":{"id":"2502.01876","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.01876","created_at":"2026-07-05T11:23:08Z"},{"alias_kind":"arxiv_version","alias_value":"2502.01876v2","created_at":"2026-07-05T11:23:08Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.01876","created_at":"2026-07-05T11:23:08Z"},{"alias_kind":"pith_short_12","alias_value":"AJR7CYNWUB5B","created_at":"2026-07-05T11:23:08Z"},{"alias_kind":"pith_short_16","alias_value":"AJR7CYNWUB5BMJ6M","created_at":"2026-07-05T11:23:08Z"},{"alias_kind":"pith_short_8","alias_value":"AJR7CYNW","created_at":"2026-07-05T11:23:08Z"}],"graph_snapshots":[{"event_id":"sha256:51ba758cdcab7c9c28bdc5314e67db4cdeeaea97154474584cff362aa6a5c857","target":"graph","created_at":"2026-07-05T11:23:08Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2502.01876/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Standard reinforcement learning (RL) assumes that an agent can observe a reward for each state-action pair. However, in practical applications, it is often difficult and costly to collect a reward for each state-action pair. While there have been several works considering RL with trajectory feedback, it is unclear if trajectory feedback is inefficient for learning when trajectories are long. In this work, we consider a model named RL with segment feedback, which offers a general paradigm filling the gap between per-state-action feedback and trajectory feedback. In this model, we consider an ep","authors_text":"Anna Winnicki, Gal Dalal, R. Srikant, Shie Mannor, Yihan Du","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-03T23:08:42Z","title":"Reinforcement Learning with Segment Feedback"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.01876","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:1e079c19e665310a558cbb7cce8254536474d2a037d21b02f0910fa9c7890c19","target":"record","created_at":"2026-07-05T11:23:08Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"951095327d0a02939cf81090632c4021aab23492e57c8f440ffb44463668f42f","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-03T23:08:42Z","title_canon_sha256":"caefdc1875359f6b5dd310e6e3b1612f58ae163913243f908e789f17172d7281"},"schema_version":"1.0","source":{"id":"2502.01876","kind":"arxiv","version":2}},"canonical_sha256":"0263f161b6a07a1627cc5a3a7a9cbb9341fc07859043fe62125b06d06025d5c2","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"0263f161b6a07a1627cc5a3a7a9cbb9341fc07859043fe62125b06d06025d5c2","first_computed_at":"2026-07-05T11:23:08.643432Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:23:08.643432Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"+43EaIq7gH/2o1s/8RLDV5kfNFc0KPQwV2QHD8m4K7cmTTvi7Lu/vU2NjDmRYm10/aKERQjG+vP5opJkgAMOAw==","signature_status":"signed_v1","signed_at":"2026-07-05T11:23:08.643959Z","signed_message":"canonical_sha256_bytes"},"source_id":"2502.01876","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:1e079c19e665310a558cbb7cce8254536474d2a037d21b02f0910fa9c7890c19","sha256:51ba758cdcab7c9c28bdc5314e67db4cdeeaea97154474584cff362aa6a5c857"],"state_sha256":"2efd776efcde55fe62ef55b5b30a15fdbccd674d35a278842c099ddb05840c24"}