{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2ERQWFQ2HOPVAQMQ4H5NWZINGH","short_pith_number":"pith:2ERQWFQ2","schema_version":"1.0","canonical_sha256":"d1230b161a3b9f504190e1fadb650d31ebd49a0488f2b0cb46159997c7114864","source":{"kind":"arxiv","id":"2408.11791","version":1},"attestation_state":"computed","paper":{"title":"Critique-out-Loud Reward Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Brandon Cui, Jonathan D. Chang, Mansheej Paul, Prithviraj Ammanabrolu, Zachary Ankner","submitted_at":"2024-08-21T17:24:15Z","abstract_excerpt":"Traditionally, reward models used for reinforcement learning from human feedback (RLHF) are trained to directly predict preference scores without leveraging the generation capabilities of the underlying large language model (LLM). This limits the capabilities of reward models as they must reason implicitly about the quality of a response, i.e., preference modeling must be performed in a single forward pass through the model. To enable reward models to reason explicitly about the quality of a response, we introduce Critique-out-Loud (CLoud) reward models. CLoud reward models operate by first ge"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.11791","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-08-21T17:24:15Z","cross_cats_sorted":[],"title_canon_sha256":"5e98832ec433f7e8bfb67621289ff5419be3a8d0708d14abe63f3091b316f911","abstract_canon_sha256":"7826dd691b965e68825521245185df5ea92e062374ddfe5eb79b948f6b07915e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:57:49.153822Z","signature_b64":"N1TFtt5vvJH1ArT9Ltu/om90Qnmh3P/VlCpathAKHi/PMaQkapD/XcYMZq7d+drecZ8e4sFRn2wFFLQi79fyBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d1230b161a3b9f504190e1fadb650d31ebd49a0488f2b0cb46159997c7114864","last_reissued_at":"2026-07-05T08:57:49.153412Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:57:49.153412Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Critique-out-Loud Reward Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Brandon Cui, Jonathan D. Chang, Mansheej Paul, Prithviraj Ammanabrolu, Zachary Ankner","submitted_at":"2024-08-21T17:24:15Z","abstract_excerpt":"Traditionally, reward models used for reinforcement learning from human feedback (RLHF) are trained to directly predict preference scores without leveraging the generation capabilities of the underlying large language model (LLM). This limits the capabilities of reward models as they must reason implicitly about the quality of a response, i.e., preference modeling must be performed in a single forward pass through the model. To enable reward models to reason explicitly about the quality of a response, we introduce Critique-out-Loud (CLoud) reward models. CLoud reward models operate by first ge"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.11791","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.11791/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.11791","created_at":"2026-07-05T08:57:49.153465+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.11791v1","created_at":"2026-07-05T08:57:49.153465+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.11791","created_at":"2026-07-05T08:57:49.153465+00:00"},{"alias_kind":"pith_short_12","alias_value":"2ERQWFQ2HOPV","created_at":"2026-07-05T08:57:49.153465+00:00"},{"alias_kind":"pith_short_16","alias_value":"2ERQWFQ2HOPVAQMQ","created_at":"2026-07-05T08:57:49.153465+00:00"},{"alias_kind":"pith_short_8","alias_value":"2ERQWFQ2","created_at":"2026-07-05T08:57:49.153465+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08077","citing_title":"Support Vector Rubrics: Closing the Gap Between Self-Generated and Human Rubrics","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":150,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00424","citing_title":"Weak Critics Make Strong Learners: On-Policy Critique Distillation for Scalable Oversight","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14636","citing_title":"Teaching Large Language Models When Not to Know: Learning Temporal Critique for Ex-Ante Reasoning","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2411.15594","citing_title":"A Survey on LLM-as-a-Judge","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2504.12501","citing_title":"Reinforcement Learning from Human Feedback","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18851","citing_title":"STRIDE: Learnable Stepwise Language Feedback for LLM Reasoning","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18653","citing_title":"Will It Go Viral? Grounding Micro-Video Popularity Prediction on the Open Web","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2505.20214","citing_title":"When Slower Isn't Truer: Inverse Scaling Law of Truthfulness in Multimodal Reasoning","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2506.01937","citing_title":"RewardBench 2: Advancing Reward Model Evaluation","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2510.24235","citing_title":"PaTaRM: Bridging Pairwise and Pointwise Signals via Preference-Aware Task-Adaptive Reward Modeling","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2509.08827","citing_title":"A Survey of Reinforcement Learning for Large Reasoning Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12620","citing_title":"Think Twice, Act Once: Verifier-Guided Action Selection For Embodied Agents","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08354","citing_title":"Auto-Rubric as Reward: From Implicit Preferences to Explicit Multimodal Generative Criteria","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03356","citing_title":"POSTCONDBENCH: Benchmarking Correctness and Completeness in Formal Postcondition Inference","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21718","citing_title":"Building a Precise Video Language with Human-AI Oversight","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13602","citing_title":"Reward Hacking in the Era of Large Models: Mechanisms, Emergent Misalignment, Challenges","ref_index":132,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20216","citing_title":"Text-to-Distribution Prediction with Quantile Tokens and Neighbor Context","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2ERQWFQ2HOPVAQMQ4H5NWZINGH","json":"https://pith.science/pith/2ERQWFQ2HOPVAQMQ4H5NWZINGH.json","graph_json":"https://pith.science/api/pith-number/2ERQWFQ2HOPVAQMQ4H5NWZINGH/graph.json","events_json":"https://pith.science/api/pith-number/2ERQWFQ2HOPVAQMQ4H5NWZINGH/events.json","paper":"https://pith.science/paper/2ERQWFQ2"},"agent_actions":{"view_html":"https://pith.science/pith/2ERQWFQ2HOPVAQMQ4H5NWZINGH","download_json":"https://pith.science/pith/2ERQWFQ2HOPVAQMQ4H5NWZINGH.json","view_paper":"https://pith.science/paper/2ERQWFQ2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.11791&json=true","fetch_graph":"https://pith.science/api/pith-number/2ERQWFQ2HOPVAQMQ4H5NWZINGH/graph.json","fetch_events":"https://pith.science/api/pith-number/2ERQWFQ2HOPVAQMQ4H5NWZINGH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2ERQWFQ2HOPVAQMQ4H5NWZINGH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2ERQWFQ2HOPVAQMQ4H5NWZINGH/action/storage_attestation","attest_author":"https://pith.science/pith/2ERQWFQ2HOPVAQMQ4H5NWZINGH/action/author_attestation","sign_citation":"https://pith.science/pith/2ERQWFQ2HOPVAQMQ4H5NWZINGH/action/citation_signature","submit_replication":"https://pith.science/pith/2ERQWFQ2HOPVAQMQ4H5NWZINGH/action/replication_record"}},"created_at":"2026-07-05T08:57:49.153465+00:00","updated_at":"2026-07-05T08:57:49.153465+00:00"}