{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:I2XVVMQN7PRT5FTC3M6TFTQHBA","short_pith_number":"pith:I2XVVMQN","schema_version":"1.0","canonical_sha256":"46af5ab20dfbe33e9662db3d32ce070812a8d26f56069c15a5eaeed620e07cf1","source":{"kind":"arxiv","id":"2503.23687","version":1},"attestation_state":"computed","paper":{"title":"MKA: Leveraging Cross-Lingual Consensus for Model Abstention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Sharad Duwal","submitted_at":"2025-03-31T03:38:12Z","abstract_excerpt":"Reliability of LLMs is questionable even as they get better at more tasks. A wider adoption of LLMs is contingent on whether they are usably factual. And if they are not, on whether they can properly calibrate their confidence in their responses. This work focuses on utilizing the multilingual knowledge of an LLM to inform its decision to abstain or answer when prompted. We develop a multilingual pipeline to calibrate the model's confidence and let it abstain when uncertain. We run several multilingual models through the pipeline to profile them across different languages. We find that the per"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.23687","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-31T03:38:12Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"1518ea67c180aaaeea53487e2e9543383fb990a185dfba1799f481ca7e4a95f0","abstract_canon_sha256":"0bbd6f28f9f624f2823629f9aeffd15ddacd5084d307eee79b220827f7434348"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:41:57.849461Z","signature_b64":"QGHb2N6FNCFgisHKh9yVz7FDEVIKQ/R53sB++32VQ2p0FH7HAvUhNxIXcXW3A6nREg0H1OeJOjGFscQyBUvlBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"46af5ab20dfbe33e9662db3d32ce070812a8d26f56069c15a5eaeed620e07cf1","last_reissued_at":"2026-07-05T10:41:57.849038Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:41:57.849038Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MKA: Leveraging Cross-Lingual Consensus for Model Abstention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Sharad Duwal","submitted_at":"2025-03-31T03:38:12Z","abstract_excerpt":"Reliability of LLMs is questionable even as they get better at more tasks. A wider adoption of LLMs is contingent on whether they are usably factual. And if they are not, on whether they can properly calibrate their confidence in their responses. This work focuses on utilizing the multilingual knowledge of an LLM to inform its decision to abstain or answer when prompted. We develop a multilingual pipeline to calibrate the model's confidence and let it abstain when uncertain. We run several multilingual models through the pipeline to profile them across different languages. We find that the per"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.23687","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.23687/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.23687","created_at":"2026-07-05T10:41:57.849093+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.23687v1","created_at":"2026-07-05T10:41:57.849093+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.23687","created_at":"2026-07-05T10:41:57.849093+00:00"},{"alias_kind":"pith_short_12","alias_value":"I2XVVMQN7PRT","created_at":"2026-07-05T10:41:57.849093+00:00"},{"alias_kind":"pith_short_16","alias_value":"I2XVVMQN7PRT5FTC","created_at":"2026-07-05T10:41:57.849093+00:00"},{"alias_kind":"pith_short_8","alias_value":"I2XVVMQN","created_at":"2026-07-05T10:41:57.849093+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.02626","citing_title":"Understanding New-Knowledge-Induced Factual Hallucinations in LLMs: Analysis and Interpretation","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I2XVVMQN7PRT5FTC3M6TFTQHBA","json":"https://pith.science/pith/I2XVVMQN7PRT5FTC3M6TFTQHBA.json","graph_json":"https://pith.science/api/pith-number/I2XVVMQN7PRT5FTC3M6TFTQHBA/graph.json","events_json":"https://pith.science/api/pith-number/I2XVVMQN7PRT5FTC3M6TFTQHBA/events.json","paper":"https://pith.science/paper/I2XVVMQN"},"agent_actions":{"view_html":"https://pith.science/pith/I2XVVMQN7PRT5FTC3M6TFTQHBA","download_json":"https://pith.science/pith/I2XVVMQN7PRT5FTC3M6TFTQHBA.json","view_paper":"https://pith.science/paper/I2XVVMQN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.23687&json=true","fetch_graph":"https://pith.science/api/pith-number/I2XVVMQN7PRT5FTC3M6TFTQHBA/graph.json","fetch_events":"https://pith.science/api/pith-number/I2XVVMQN7PRT5FTC3M6TFTQHBA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I2XVVMQN7PRT5FTC3M6TFTQHBA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I2XVVMQN7PRT5FTC3M6TFTQHBA/action/storage_attestation","attest_author":"https://pith.science/pith/I2XVVMQN7PRT5FTC3M6TFTQHBA/action/author_attestation","sign_citation":"https://pith.science/pith/I2XVVMQN7PRT5FTC3M6TFTQHBA/action/citation_signature","submit_replication":"https://pith.science/pith/I2XVVMQN7PRT5FTC3M6TFTQHBA/action/replication_record"}},"created_at":"2026-07-05T10:41:57.849093+00:00","updated_at":"2026-07-05T10:41:57.849093+00:00"}