{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YTFS5M7KZVVCYTA556C5ICUJQV","short_pith_number":"pith:YTFS5M7K","schema_version":"1.0","canonical_sha256":"c4cb2eb3eacd6a2c4c1def85d40a89854a827a8b156022b9898cc472a9a217f7","source":{"kind":"arxiv","id":"2410.04166","version":3},"attestation_state":"computed","paper":{"title":"Learning from negative feedback, or positive feedback or both","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Abbas Abdolmaleki, Bilal Piot, Bobak Shahriari, Jonas Buchli, Jost Tobias Springenberg, Junhyuk Oh, Martin Riedmiller, Michael Bloesch, Nicolas Heess, Rishabh Joshi, Thomas Lampe, Tim Hertweck","submitted_at":"2024-10-05T14:04:03Z","abstract_excerpt":"Existing preference optimization methods often assume scenarios where paired preference feedback (preferred/positive vs. dis-preferred/negative examples) is available. This requirement limits their applicability in scenarios where only unpaired feedback--for example, either positive or negative--is available. To address this, we introduce a novel approach that decouples learning from positive and negative feedback. This decoupling enables control over the influence of each feedback type and, importantly, allows learning even when only one feedback type is present. A key contribution is demonst"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.04166","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-05T14:04:03Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"657ae31776bdbf2fc7f249c7a02aeed5673186c66b3c5d5a5cd269e0ae97a4fd","abstract_canon_sha256":"f6f98c87cd35773cb21daeb2545182d50346670ea8c6c4ee89c3ac8937ed4601"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:26:08.391912Z","signature_b64":"Q8XAoQWm9uPP2K3Q4YBp8HWO/k8On+7fqORDImGOcOD49D4vhW9vnfhQ7Ab6qZlpl1g029tc4klN6elzZ0eZDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c4cb2eb3eacd6a2c4c1def85d40a89854a827a8b156022b9898cc472a9a217f7","last_reissued_at":"2026-07-05T10:26:08.391421Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:26:08.391421Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning from negative feedback, or positive feedback or both","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Abbas Abdolmaleki, Bilal Piot, Bobak Shahriari, Jonas Buchli, Jost Tobias Springenberg, Junhyuk Oh, Martin Riedmiller, Michael Bloesch, Nicolas Heess, Rishabh Joshi, Thomas Lampe, Tim Hertweck","submitted_at":"2024-10-05T14:04:03Z","abstract_excerpt":"Existing preference optimization methods often assume scenarios where paired preference feedback (preferred/positive vs. dis-preferred/negative examples) is available. This requirement limits their applicability in scenarios where only unpaired feedback--for example, either positive or negative--is available. To address this, we introduce a novel approach that decouples learning from positive and negative feedback. This decoupling enables control over the influence of each feedback type and, importantly, allows learning even when only one feedback type is present. A key contribution is demonst"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.04166","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.04166/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.04166","created_at":"2026-07-05T10:26:08.391482+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.04166v3","created_at":"2026-07-05T10:26:08.391482+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.04166","created_at":"2026-07-05T10:26:08.391482+00:00"},{"alias_kind":"pith_short_12","alias_value":"YTFS5M7KZVVC","created_at":"2026-07-05T10:26:08.391482+00:00"},{"alias_kind":"pith_short_16","alias_value":"YTFS5M7KZVVCYTA5","created_at":"2026-07-05T10:26:08.391482+00:00"},{"alias_kind":"pith_short_8","alias_value":"YTFS5M7K","created_at":"2026-07-05T10:26:08.391482+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.12856","citing_title":"Supervised Fine Tuning on Curated Data is Reinforcement Learning (and can be improved)","ref_index":10,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YTFS5M7KZVVCYTA556C5ICUJQV","json":"https://pith.science/pith/YTFS5M7KZVVCYTA556C5ICUJQV.json","graph_json":"https://pith.science/api/pith-number/YTFS5M7KZVVCYTA556C5ICUJQV/graph.json","events_json":"https://pith.science/api/pith-number/YTFS5M7KZVVCYTA556C5ICUJQV/events.json","paper":"https://pith.science/paper/YTFS5M7K"},"agent_actions":{"view_html":"https://pith.science/pith/YTFS5M7KZVVCYTA556C5ICUJQV","download_json":"https://pith.science/pith/YTFS5M7KZVVCYTA556C5ICUJQV.json","view_paper":"https://pith.science/paper/YTFS5M7K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.04166&json=true","fetch_graph":"https://pith.science/api/pith-number/YTFS5M7KZVVCYTA556C5ICUJQV/graph.json","fetch_events":"https://pith.science/api/pith-number/YTFS5M7KZVVCYTA556C5ICUJQV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YTFS5M7KZVVCYTA556C5ICUJQV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YTFS5M7KZVVCYTA556C5ICUJQV/action/storage_attestation","attest_author":"https://pith.science/pith/YTFS5M7KZVVCYTA556C5ICUJQV/action/author_attestation","sign_citation":"https://pith.science/pith/YTFS5M7KZVVCYTA556C5ICUJQV/action/citation_signature","submit_replication":"https://pith.science/pith/YTFS5M7KZVVCYTA556C5ICUJQV/action/replication_record"}},"created_at":"2026-07-05T10:26:08.391482+00:00","updated_at":"2026-07-05T10:26:08.391482+00:00"}