{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:7XIWNVVK5XSJEGV252Y7WDYEMQ","short_pith_number":"pith:7XIWNVVK","canonical_record":{"source":{"id":"2504.05294","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-04-07T17:49:23Z","cross_cats_sorted":[],"title_canon_sha256":"092f4fbc285166457daf82e1a30c95a4bbd3403964667f2c206bd845f5f4d698","abstract_canon_sha256":"2a55f121cf426c8b9e488bf508202b534d9adb9b2c65de89da70b7adbfd3cd00"},"schema_version":"1.0"},"canonical_sha256":"fdd166d6aaede4921abaeeb1fb0f04641270422dd51f300bbf180504bd007269","source":{"kind":"arxiv","id":"2504.05294","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2504.05294","created_at":"2026-07-05T11:37:17Z"},{"alias_kind":"arxiv_version","alias_value":"2504.05294v2","created_at":"2026-07-05T11:37:17Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.05294","created_at":"2026-07-05T11:37:17Z"},{"alias_kind":"pith_short_12","alias_value":"7XIWNVVK5XSJ","created_at":"2026-07-05T11:37:17Z"},{"alias_kind":"pith_short_16","alias_value":"7XIWNVVK5XSJEGV2","created_at":"2026-07-05T11:37:17Z"},{"alias_kind":"pith_short_8","alias_value":"7XIWNVVK","created_at":"2026-07-05T11:37:17Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:7XIWNVVK5XSJEGV252Y7WDYEMQ","target":"record","payload":{"canonical_record":{"source":{"id":"2504.05294","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-04-07T17:49:23Z","cross_cats_sorted":[],"title_canon_sha256":"092f4fbc285166457daf82e1a30c95a4bbd3403964667f2c206bd845f5f4d698","abstract_canon_sha256":"2a55f121cf426c8b9e488bf508202b534d9adb9b2c65de89da70b7adbfd3cd00"},"schema_version":"1.0"},"canonical_sha256":"fdd166d6aaede4921abaeeb1fb0f04641270422dd51f300bbf180504bd007269","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:37:17.867677Z","signature_b64":"Z48f83Qd69qEdHgtkDeTlLCPtJKk5Wkt6AzAHvM9AOadFxTOGO19Uaqs35BsX9Vza0mrKGDN7BCvin3BU5MJDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fdd166d6aaede4921abaeeb1fb0f04641270422dd51f300bbf180504bd007269","last_reissued_at":"2026-07-05T11:37:17.867202Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:37:17.867202Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2504.05294","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:37:17Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"8MHt1MA5qxS7Ew+uMNz2M+H5zS88wTKRoaLe9s0oAoLWZjAhkfk6Bnc34doqX1ahEyQUfSGOP3P/TLJZuXyNDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-16T18:57:14.035356Z"},"content_sha256":"3396fc310fb1b9660294db84de18898fd0c3cb1ccbb42d9e82cc82bb144cc7cb","schema_version":"1.0","event_id":"sha256:3396fc310fb1b9660294db84de18898fd0c3cb1ccbb42d9e82cc82bb144cc7cb"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:7XIWNVVK5XSJEGV252Y7WDYEMQ","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Truthful or Fabricated? Using Causal Attribution to Mitigate Reward Hacking in Explanations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ivan Titov, Pedro Ferreira, Wilker Aziz","submitted_at":"2025-04-07T17:49:23Z","abstract_excerpt":"Chain-of-thought explanations are widely used to inspect the decision process of large language models (LLMs) and to evaluate the trustworthiness of model outputs, making them important for effective collaboration between LLMs and humans. We demonstrate that preference optimization - a key step in the alignment phase - can inadvertently reduce the faithfulness of these explanations. This occurs because the reward model (RM), which guides alignment, is tasked with optimizing both the expected quality of the response and the appropriateness of the explanations (e.g., minimizing bias or adhering "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.05294","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.05294/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:37:17Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"bQ5bUvigAe3uwOMJqW7QWggdmweCztIMlnQA0tfpE3RUKPnDitLovTwy3kJQoamzYLV62bM0AvVKFhp88x4bDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-16T18:57:14.036274Z"},"content_sha256":"4e2bcce742ecb434d4c204d5b590c0473ba6ba9c6a3e9d4ec2873f4e60dea9a1","schema_version":"1.0","event_id":"sha256:4e2bcce742ecb434d4c204d5b590c0473ba6ba9c6a3e9d4ec2873f4e60dea9a1"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/7XIWNVVK5XSJEGV252Y7WDYEMQ/bundle.json","state_url":"https://pith.science/pith/7XIWNVVK5XSJEGV252Y7WDYEMQ/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/7XIWNVVK5XSJEGV252Y7WDYEMQ/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-16T18:57:14Z","links":{"resolver":"https://pith.science/pith/7XIWNVVK5XSJEGV252Y7WDYEMQ","bundle":"https://pith.science/pith/7XIWNVVK5XSJEGV252Y7WDYEMQ/bundle.json","state":"https://pith.science/pith/7XIWNVVK5XSJEGV252Y7WDYEMQ/state.json","well_known_bundle":"https://pith.science/.well-known/pith/7XIWNVVK5XSJEGV252Y7WDYEMQ/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:7XIWNVVK5XSJEGV252Y7WDYEMQ","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"2a55f121cf426c8b9e488bf508202b534d9adb9b2c65de89da70b7adbfd3cd00","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-04-07T17:49:23Z","title_canon_sha256":"092f4fbc285166457daf82e1a30c95a4bbd3403964667f2c206bd845f5f4d698"},"schema_version":"1.0","source":{"id":"2504.05294","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2504.05294","created_at":"2026-07-05T11:37:17Z"},{"alias_kind":"arxiv_version","alias_value":"2504.05294v2","created_at":"2026-07-05T11:37:17Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.05294","created_at":"2026-07-05T11:37:17Z"},{"alias_kind":"pith_short_12","alias_value":"7XIWNVVK5XSJ","created_at":"2026-07-05T11:37:17Z"},{"alias_kind":"pith_short_16","alias_value":"7XIWNVVK5XSJEGV2","created_at":"2026-07-05T11:37:17Z"},{"alias_kind":"pith_short_8","alias_value":"7XIWNVVK","created_at":"2026-07-05T11:37:17Z"}],"graph_snapshots":[{"event_id":"sha256:4e2bcce742ecb434d4c204d5b590c0473ba6ba9c6a3e9d4ec2873f4e60dea9a1","target":"graph","created_at":"2026-07-05T11:37:17Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2504.05294/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Chain-of-thought explanations are widely used to inspect the decision process of large language models (LLMs) and to evaluate the trustworthiness of model outputs, making them important for effective collaboration between LLMs and humans. We demonstrate that preference optimization - a key step in the alignment phase - can inadvertently reduce the faithfulness of these explanations. This occurs because the reward model (RM), which guides alignment, is tasked with optimizing both the expected quality of the response and the appropriateness of the explanations (e.g., minimizing bias or adhering ","authors_text":"Ivan Titov, Pedro Ferreira, Wilker Aziz","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-04-07T17:49:23Z","title":"Truthful or Fabricated? Using Causal Attribution to Mitigate Reward Hacking in Explanations"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.05294","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:3396fc310fb1b9660294db84de18898fd0c3cb1ccbb42d9e82cc82bb144cc7cb","target":"record","created_at":"2026-07-05T11:37:17Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"2a55f121cf426c8b9e488bf508202b534d9adb9b2c65de89da70b7adbfd3cd00","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-04-07T17:49:23Z","title_canon_sha256":"092f4fbc285166457daf82e1a30c95a4bbd3403964667f2c206bd845f5f4d698"},"schema_version":"1.0","source":{"id":"2504.05294","kind":"arxiv","version":2}},"canonical_sha256":"fdd166d6aaede4921abaeeb1fb0f04641270422dd51f300bbf180504bd007269","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"fdd166d6aaede4921abaeeb1fb0f04641270422dd51f300bbf180504bd007269","first_computed_at":"2026-07-05T11:37:17.867202Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:37:17.867202Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"Z48f83Qd69qEdHgtkDeTlLCPtJKk5Wkt6AzAHvM9AOadFxTOGO19Uaqs35BsX9Vza0mrKGDN7BCvin3BU5MJDQ==","signature_status":"signed_v1","signed_at":"2026-07-05T11:37:17.867677Z","signed_message":"canonical_sha256_bytes"},"source_id":"2504.05294","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:3396fc310fb1b9660294db84de18898fd0c3cb1ccbb42d9e82cc82bb144cc7cb","sha256:4e2bcce742ecb434d4c204d5b590c0473ba6ba9c6a3e9d4ec2873f4e60dea9a1"],"state_sha256":"5279190b31778578b5c2ef5521ce7e3553258bd635fd93d6f40011409f1ac71a"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"oYud6eDL0fVkBb/2TL8umZTaWewqoC+QGn6fi4Ex/muk0XK3XAwieT3aihY1pGRNzXyp95SaYyTL4ehkTt5SCA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-16T18:57:14.062460Z","bundle_sha256":"3e6c5d72334be5fc75c92ba0ad8a45874fea6f17cd0f604c936dbb3cb9fe9c2e"}}