{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:J2Z2KQDKMKNKRF2TI3F4PLPHZA","short_pith_number":"pith:J2Z2KQDK","schema_version":"1.0","canonical_sha256":"4eb3a5406a629aa8975346cbc7ade7c80d863413feb6890b46e6d6e3cea668c6","source":{"kind":"arxiv","id":"2410.11302","version":1},"attestation_state":"computed","paper":{"title":"Have the VLMs Lost Confidence? A Study of Sycophancy in VLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Leyi Yang, Linsheng Lu, Qi Zhang, Rui Zheng, Shuo Li, Tao Gui, Tao Ji, Xiaohui Zhao, Xiaoran Fan, Xuanjing Huang, Yuming Yang, Yuran Wang, Zhiheng Xi","submitted_at":"2024-10-15T05:48:14Z","abstract_excerpt":"In the study of LLMs, sycophancy represents a prevalent hallucination that poses significant challenges to these models. Specifically, LLMs often fail to adhere to original correct responses, instead blindly agreeing with users' opinions, even when those opinions are incorrect or malicious. However, research on sycophancy in visual language models (VLMs) has been scarce. In this work, we extend the exploration of sycophancy from LLMs to VLMs, introducing the MM-SY benchmark to evaluate this phenomenon. We present evaluation results from multiple representative models, addressing the gap in syc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.11302","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-15T05:48:14Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"444e8df3c78f30ba723fe92219ce80bb70dfa355450221c84c87fca0557d80fe","abstract_canon_sha256":"9737c9a263ea611a6e4ed9dfc29e79ef494adaf18ffa22f25537aa5b3b79f744"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:20:51.395820Z","signature_b64":"ME9Z5z1fvzZ3X7LRPsaMhtB2iWjRcwoRRxEOOdGeZ4q4Dco1pDgB0FZIZCX1pxsqO4o4INXB0X75p0GNBLElCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4eb3a5406a629aa8975346cbc7ade7c80d863413feb6890b46e6d6e3cea668c6","last_reissued_at":"2026-07-05T09:20:51.395328Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:20:51.395328Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Have the VLMs Lost Confidence? A Study of Sycophancy in VLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Leyi Yang, Linsheng Lu, Qi Zhang, Rui Zheng, Shuo Li, Tao Gui, Tao Ji, Xiaohui Zhao, Xiaoran Fan, Xuanjing Huang, Yuming Yang, Yuran Wang, Zhiheng Xi","submitted_at":"2024-10-15T05:48:14Z","abstract_excerpt":"In the study of LLMs, sycophancy represents a prevalent hallucination that poses significant challenges to these models. Specifically, LLMs often fail to adhere to original correct responses, instead blindly agreeing with users' opinions, even when those opinions are incorrect or malicious. However, research on sycophancy in visual language models (VLMs) has been scarce. In this work, we extend the exploration of sycophancy from LLMs to VLMs, introducing the MM-SY benchmark to evaluate this phenomenon. We present evaluation results from multiple representative models, addressing the gap in syc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.11302","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.11302/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.11302","created_at":"2026-07-05T09:20:51.395394+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.11302v1","created_at":"2026-07-05T09:20:51.395394+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.11302","created_at":"2026-07-05T09:20:51.395394+00:00"},{"alias_kind":"pith_short_12","alias_value":"J2Z2KQDKMKNK","created_at":"2026-07-05T09:20:51.395394+00:00"},{"alias_kind":"pith_short_16","alias_value":"J2Z2KQDKMKNKRF2T","created_at":"2026-07-05T09:20:51.395394+00:00"},{"alias_kind":"pith_short_8","alias_value":"J2Z2KQDK","created_at":"2026-07-05T09:20:51.395394+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.21979","citing_title":"Benchmarking and Mitigating Sycophancy in Medical Vision Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21979","citing_title":"Benchmarking and Mitigating Sycophancy in Medical Vision Language Models","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J2Z2KQDKMKNKRF2TI3F4PLPHZA","json":"https://pith.science/pith/J2Z2KQDKMKNKRF2TI3F4PLPHZA.json","graph_json":"https://pith.science/api/pith-number/J2Z2KQDKMKNKRF2TI3F4PLPHZA/graph.json","events_json":"https://pith.science/api/pith-number/J2Z2KQDKMKNKRF2TI3F4PLPHZA/events.json","paper":"https://pith.science/paper/J2Z2KQDK"},"agent_actions":{"view_html":"https://pith.science/pith/J2Z2KQDKMKNKRF2TI3F4PLPHZA","download_json":"https://pith.science/pith/J2Z2KQDKMKNKRF2TI3F4PLPHZA.json","view_paper":"https://pith.science/paper/J2Z2KQDK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.11302&json=true","fetch_graph":"https://pith.science/api/pith-number/J2Z2KQDKMKNKRF2TI3F4PLPHZA/graph.json","fetch_events":"https://pith.science/api/pith-number/J2Z2KQDKMKNKRF2TI3F4PLPHZA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J2Z2KQDKMKNKRF2TI3F4PLPHZA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J2Z2KQDKMKNKRF2TI3F4PLPHZA/action/storage_attestation","attest_author":"https://pith.science/pith/J2Z2KQDKMKNKRF2TI3F4PLPHZA/action/author_attestation","sign_citation":"https://pith.science/pith/J2Z2KQDKMKNKRF2TI3F4PLPHZA/action/citation_signature","submit_replication":"https://pith.science/pith/J2Z2KQDKMKNKRF2TI3F4PLPHZA/action/replication_record"}},"created_at":"2026-07-05T09:20:51.395394+00:00","updated_at":"2026-07-05T09:20:51.395394+00:00"}