{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XG46OGGNURIF32TMVBC4E6FJWM","short_pith_number":"pith:XG46OGGN","schema_version":"1.0","canonical_sha256":"b9b9e718cda4505dea6ca845c278a9b323e20f443a7928b3d3daa91d4ac1691d","source":{"kind":"arxiv","id":"2505.13488","version":1},"attestation_state":"computed","paper":{"title":"Source framing triggers systematic evaluation bias in Large Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.CL","authors_text":"Federico Germani, Giovanni Spitale","submitted_at":"2025-05-14T07:42:27Z","abstract_excerpt":"Large Language Models (LLMs) are increasingly used not only to generate text but also to evaluate it, raising urgent questions about whether their judgments are consistent, unbiased, and robust to framing effects. In this study, we systematically examine inter- and intra-model agreement across four state-of-the-art LLMs (OpenAI o3-mini, Deepseek Reasoner, xAI Grok 2, and Mistral) tasked with evaluating 4,800 narrative statements on 24 different topics of social, political, and public health relevance, for a total of 192,000 assessments. We manipulate the disclosed source of each statement to a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.13488","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-14T07:42:27Z","cross_cats_sorted":["cs.CY"],"title_canon_sha256":"8c7cc6518d828451e927697d830019944f7119ed70f8b0efabc55c565e86d285","abstract_canon_sha256":"9650d1168cbbfa53733f12029abf2b5101f8246fe8147a120c9e59e53b268a4b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:05:46.611046Z","signature_b64":"5h2YZASIvSzu0QB8wV2d6hW4lBDE0eAMWrQvr1fwjm1aUzAL5OP1lv7fzAzgTSV8O/I2fXmJzHAuSMxEGZxWBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b9b9e718cda4505dea6ca845c278a9b323e20f443a7928b3d3daa91d4ac1691d","last_reissued_at":"2026-07-05T11:05:46.610559Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:05:46.610559Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Source framing triggers systematic evaluation bias in Large Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.CL","authors_text":"Federico Germani, Giovanni Spitale","submitted_at":"2025-05-14T07:42:27Z","abstract_excerpt":"Large Language Models (LLMs) are increasingly used not only to generate text but also to evaluate it, raising urgent questions about whether their judgments are consistent, unbiased, and robust to framing effects. In this study, we systematically examine inter- and intra-model agreement across four state-of-the-art LLMs (OpenAI o3-mini, Deepseek Reasoner, xAI Grok 2, and Mistral) tasked with evaluating 4,800 narrative statements on 24 different topics of social, political, and public health relevance, for a total of 192,000 assessments. We manipulate the disclosed source of each statement to a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.13488","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.13488/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.13488","created_at":"2026-07-05T11:05:46.610623+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.13488v1","created_at":"2026-07-05T11:05:46.610623+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.13488","created_at":"2026-07-05T11:05:46.610623+00:00"},{"alias_kind":"pith_short_12","alias_value":"XG46OGGNURIF","created_at":"2026-07-05T11:05:46.610623+00:00"},{"alias_kind":"pith_short_16","alias_value":"XG46OGGNURIF32TM","created_at":"2026-07-05T11:05:46.610623+00:00"},{"alias_kind":"pith_short_8","alias_value":"XG46OGGN","created_at":"2026-07-05T11:05:46.610623+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.10528","citing_title":"Collective Alignment in LLM Multi-Agent Systems: Disentangling Bias from Cooperation via Statistical Physics","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XG46OGGNURIF32TMVBC4E6FJWM","json":"https://pith.science/pith/XG46OGGNURIF32TMVBC4E6FJWM.json","graph_json":"https://pith.science/api/pith-number/XG46OGGNURIF32TMVBC4E6FJWM/graph.json","events_json":"https://pith.science/api/pith-number/XG46OGGNURIF32TMVBC4E6FJWM/events.json","paper":"https://pith.science/paper/XG46OGGN"},"agent_actions":{"view_html":"https://pith.science/pith/XG46OGGNURIF32TMVBC4E6FJWM","download_json":"https://pith.science/pith/XG46OGGNURIF32TMVBC4E6FJWM.json","view_paper":"https://pith.science/paper/XG46OGGN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.13488&json=true","fetch_graph":"https://pith.science/api/pith-number/XG46OGGNURIF32TMVBC4E6FJWM/graph.json","fetch_events":"https://pith.science/api/pith-number/XG46OGGNURIF32TMVBC4E6FJWM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XG46OGGNURIF32TMVBC4E6FJWM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XG46OGGNURIF32TMVBC4E6FJWM/action/storage_attestation","attest_author":"https://pith.science/pith/XG46OGGNURIF32TMVBC4E6FJWM/action/author_attestation","sign_citation":"https://pith.science/pith/XG46OGGNURIF32TMVBC4E6FJWM/action/citation_signature","submit_replication":"https://pith.science/pith/XG46OGGNURIF32TMVBC4E6FJWM/action/replication_record"}},"created_at":"2026-07-05T11:05:46.610623+00:00","updated_at":"2026-07-05T11:05:46.610623+00:00"}