{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:AJR7CYNWUB5BMJ6MLI5HVHF3SN","short_pith_number":"pith:AJR7CYNW","schema_version":"1.0","canonical_sha256":"0263f161b6a07a1627cc5a3a7a9cbb9341fc07859043fe62125b06d06025d5c2","source":{"kind":"arxiv","id":"2502.01876","version":2},"attestation_state":"computed","paper":{"title":"Reinforcement Learning with Segment Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Anna Winnicki, Gal Dalal, R. Srikant, Shie Mannor, Yihan Du","submitted_at":"2025-02-03T23:08:42Z","abstract_excerpt":"Standard reinforcement learning (RL) assumes that an agent can observe a reward for each state-action pair. However, in practical applications, it is often difficult and costly to collect a reward for each state-action pair. While there have been several works considering RL with trajectory feedback, it is unclear if trajectory feedback is inefficient for learning when trajectories are long. In this work, we consider a model named RL with segment feedback, which offers a general paradigm filling the gap between per-state-action feedback and trajectory feedback. In this model, we consider an ep"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.01876","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-03T23:08:42Z","cross_cats_sorted":[],"title_canon_sha256":"caefdc1875359f6b5dd310e6e3b1612f58ae163913243f908e789f17172d7281","abstract_canon_sha256":"951095327d0a02939cf81090632c4021aab23492e57c8f440ffb44463668f42f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:23:08.643959Z","signature_b64":"+43EaIq7gH/2o1s/8RLDV5kfNFc0KPQwV2QHD8m4K7cmTTvi7Lu/vU2NjDmRYm10/aKERQjG+vP5opJkgAMOAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0263f161b6a07a1627cc5a3a7a9cbb9341fc07859043fe62125b06d06025d5c2","last_reissued_at":"2026-07-05T11:23:08.643432Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:23:08.643432Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reinforcement Learning with Segment Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Anna Winnicki, Gal Dalal, R. Srikant, Shie Mannor, Yihan Du","submitted_at":"2025-02-03T23:08:42Z","abstract_excerpt":"Standard reinforcement learning (RL) assumes that an agent can observe a reward for each state-action pair. However, in practical applications, it is often difficult and costly to collect a reward for each state-action pair. While there have been several works considering RL with trajectory feedback, it is unclear if trajectory feedback is inefficient for learning when trajectories are long. In this work, we consider a model named RL with segment feedback, which offers a general paradigm filling the gap between per-state-action feedback and trajectory feedback. In this model, we consider an ep"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.01876","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.01876/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.01876","created_at":"2026-07-05T11:23:08.643488+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.01876v2","created_at":"2026-07-05T11:23:08.643488+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.01876","created_at":"2026-07-05T11:23:08.643488+00:00"},{"alias_kind":"pith_short_12","alias_value":"AJR7CYNWUB5B","created_at":"2026-07-05T11:23:08.643488+00:00"},{"alias_kind":"pith_short_16","alias_value":"AJR7CYNWUB5BMJ6M","created_at":"2026-07-05T11:23:08.643488+00:00"},{"alias_kind":"pith_short_8","alias_value":"AJR7CYNW","created_at":"2026-07-05T11:23:08.643488+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.02951","citing_title":"SP3O: Reinforcement Learning from Segment Preferences without Reward Modeling","ref_index":17,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AJR7CYNWUB5BMJ6MLI5HVHF3SN","json":"https://pith.science/pith/AJR7CYNWUB5BMJ6MLI5HVHF3SN.json","graph_json":"https://pith.science/api/pith-number/AJR7CYNWUB5BMJ6MLI5HVHF3SN/graph.json","events_json":"https://pith.science/api/pith-number/AJR7CYNWUB5BMJ6MLI5HVHF3SN/events.json","paper":"https://pith.science/paper/AJR7CYNW"},"agent_actions":{"view_html":"https://pith.science/pith/AJR7CYNWUB5BMJ6MLI5HVHF3SN","download_json":"https://pith.science/pith/AJR7CYNWUB5BMJ6MLI5HVHF3SN.json","view_paper":"https://pith.science/paper/AJR7CYNW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.01876&json=true","fetch_graph":"https://pith.science/api/pith-number/AJR7CYNWUB5BMJ6MLI5HVHF3SN/graph.json","fetch_events":"https://pith.science/api/pith-number/AJR7CYNWUB5BMJ6MLI5HVHF3SN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AJR7CYNWUB5BMJ6MLI5HVHF3SN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AJR7CYNWUB5BMJ6MLI5HVHF3SN/action/storage_attestation","attest_author":"https://pith.science/pith/AJR7CYNWUB5BMJ6MLI5HVHF3SN/action/author_attestation","sign_citation":"https://pith.science/pith/AJR7CYNWUB5BMJ6MLI5HVHF3SN/action/citation_signature","submit_replication":"https://pith.science/pith/AJR7CYNWUB5BMJ6MLI5HVHF3SN/action/replication_record"}},"created_at":"2026-07-05T11:23:08.643488+00:00","updated_at":"2026-07-05T11:23:08.643488+00:00"}