{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QTDSRWLCRD5U4J3TWPVDXBJKUV","short_pith_number":"pith:QTDSRWLC","schema_version":"1.0","canonical_sha256":"84c728d96288fb4e2773b3ea3b852aa578c04cdc572f8790d4faccbbba75f3b3","source":{"kind":"arxiv","id":"2506.11111","version":2},"attestation_state":"computed","paper":{"title":"Evaluating and Improving Robustness in Large Language Models: A Survey and Future Directions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dacao Zhang, Guangyi Lv, Kui Yu, Kun Zhang, Le Wu","submitted_at":"2025-06-08T16:20:12Z","abstract_excerpt":"Large Language Models (LLMs) have gained enormous attention in recent years due to their capability of understanding and generating natural languages. With the rapid development and wild-range applications (e.g., Agents, Embodied Intelligence), the robustness of LLMs has received increased attention. As the core brain of many AI applications, the robustness of LLMs requires that models should not only generate consistent contents, but also ensure the correctness and stability of generated content when dealing with unexpeted application scenarios (e.g., toxic prompts, limited noise domain data,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.11111","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-08T16:20:12Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b94675cffc65a5e9f5fa30357608c5540de368e0c7fb55411702b2e8a230fe62","abstract_canon_sha256":"cf35dc4ab9cabd4d36e404e2ad19209b3a67dc84dcc2f5a6960bdf0e2b28a40e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:34:09.337492Z","signature_b64":"UbVSKXkA3lgocxbQNq7he6KH3rkl1uixQ1G3hc4LJN74AG8FMwYlxn1FKGl5St2U2GA4XUp8Ybe6TqyrJxYUCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"84c728d96288fb4e2773b3ea3b852aa578c04cdc572f8790d4faccbbba75f3b3","last_reissued_at":"2026-07-05T11:34:09.336979Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:34:09.336979Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating and Improving Robustness in Large Language Models: A Survey and Future Directions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dacao Zhang, Guangyi Lv, Kui Yu, Kun Zhang, Le Wu","submitted_at":"2025-06-08T16:20:12Z","abstract_excerpt":"Large Language Models (LLMs) have gained enormous attention in recent years due to their capability of understanding and generating natural languages. With the rapid development and wild-range applications (e.g., Agents, Embodied Intelligence), the robustness of LLMs has received increased attention. As the core brain of many AI applications, the robustness of LLMs requires that models should not only generate consistent contents, but also ensure the correctness and stability of generated content when dealing with unexpeted application scenarios (e.g., toxic prompts, limited noise domain data,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.11111","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.11111/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.11111","created_at":"2026-07-05T11:34:09.337039+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.11111v2","created_at":"2026-07-05T11:34:09.337039+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.11111","created_at":"2026-07-05T11:34:09.337039+00:00"},{"alias_kind":"pith_short_12","alias_value":"QTDSRWLCRD5U","created_at":"2026-07-05T11:34:09.337039+00:00"},{"alias_kind":"pith_short_16","alias_value":"QTDSRWLCRD5U4J3T","created_at":"2026-07-05T11:34:09.337039+00:00"},{"alias_kind":"pith_short_8","alias_value":"QTDSRWLC","created_at":"2026-07-05T11:34:09.337039+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12731","citing_title":"Normative Robustness as a Frontier for Non-Verifiable Reasoning in LLMs","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24213","citing_title":"Towards Evaluation Engineering: An Empirical Study of ML Evaluation Harnesses in the Wild","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26795","citing_title":"What Does Chain-of-Thought Contribute at Probe Time? Evidence for Local Co-Occurrence Activation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15393","citing_title":"LPDS: Evaluating LLM Robustness Through Logic-Preserving Difficulty Scaling","ref_index":65,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QTDSRWLCRD5U4J3TWPVDXBJKUV","json":"https://pith.science/pith/QTDSRWLCRD5U4J3TWPVDXBJKUV.json","graph_json":"https://pith.science/api/pith-number/QTDSRWLCRD5U4J3TWPVDXBJKUV/graph.json","events_json":"https://pith.science/api/pith-number/QTDSRWLCRD5U4J3TWPVDXBJKUV/events.json","paper":"https://pith.science/paper/QTDSRWLC"},"agent_actions":{"view_html":"https://pith.science/pith/QTDSRWLCRD5U4J3TWPVDXBJKUV","download_json":"https://pith.science/pith/QTDSRWLCRD5U4J3TWPVDXBJKUV.json","view_paper":"https://pith.science/paper/QTDSRWLC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.11111&json=true","fetch_graph":"https://pith.science/api/pith-number/QTDSRWLCRD5U4J3TWPVDXBJKUV/graph.json","fetch_events":"https://pith.science/api/pith-number/QTDSRWLCRD5U4J3TWPVDXBJKUV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QTDSRWLCRD5U4J3TWPVDXBJKUV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QTDSRWLCRD5U4J3TWPVDXBJKUV/action/storage_attestation","attest_author":"https://pith.science/pith/QTDSRWLCRD5U4J3TWPVDXBJKUV/action/author_attestation","sign_citation":"https://pith.science/pith/QTDSRWLCRD5U4J3TWPVDXBJKUV/action/citation_signature","submit_replication":"https://pith.science/pith/QTDSRWLCRD5U4J3TWPVDXBJKUV/action/replication_record"}},"created_at":"2026-07-05T11:34:09.337039+00:00","updated_at":"2026-07-05T11:34:09.337039+00:00"}