{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3FCHIDRQ7EGDRCSTUYJNMA52NG","short_pith_number":"pith:3FCHIDRQ","schema_version":"1.0","canonical_sha256":"d944740e30f90c388a53a612d603ba69a0b8e4754dc99f51afed7da3dabb49a0","source":{"kind":"arxiv","id":"2410.01208","version":3},"attestation_state":"computed","paper":{"title":"StringLLM: Understanding the String Processing Capability of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hao Fu, Jindong Wang, Neil Zhenqiang Gong, Xilong Wang","submitted_at":"2024-10-02T03:23:29Z","abstract_excerpt":"String processing, which mainly involves the analysis and manipulation of strings, is a fundamental component of modern computing. Despite the significant advancements of large language models (LLMs) in various natural language processing (NLP) tasks, their capability in string processing remains underexplored and underdeveloped. To bridge this gap, we present a comprehensive study of LLMs' string processing capability. In particular, we first propose StringLLM, a method to construct datasets for benchmarking string processing capability of LLMs. We use StringLLM to build a series of datasets,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.01208","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-02T03:23:29Z","cross_cats_sorted":[],"title_canon_sha256":"2feef5e01dc0f324b28dd6b3fad4062cf646c553757c9007afdb6807c18a235d","abstract_canon_sha256":"44685e62e2b9c006862d8a0e56cd270dafd9125883ee1a6981d68b73850a4032"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:05:13.088500Z","signature_b64":"7OWtsvOa5N2gIpnM5f5tZCek2PA7LWd3BpRHKrnWjCeQXJFCDhoKk/jNhMj3eyrPiiWRunGzgoycHfvGmwJVBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d944740e30f90c388a53a612d603ba69a0b8e4754dc99f51afed7da3dabb49a0","last_reissued_at":"2026-07-05T10:05:13.088020Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:05:13.088020Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"StringLLM: Understanding the String Processing Capability of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hao Fu, Jindong Wang, Neil Zhenqiang Gong, Xilong Wang","submitted_at":"2024-10-02T03:23:29Z","abstract_excerpt":"String processing, which mainly involves the analysis and manipulation of strings, is a fundamental component of modern computing. Despite the significant advancements of large language models (LLMs) in various natural language processing (NLP) tasks, their capability in string processing remains underexplored and underdeveloped. To bridge this gap, we present a comprehensive study of LLMs' string processing capability. In particular, we first propose StringLLM, a method to construct datasets for benchmarking string processing capability of LLMs. We use StringLLM to build a series of datasets,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.01208","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.01208/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.01208","created_at":"2026-07-05T10:05:13.088075+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.01208v3","created_at":"2026-07-05T10:05:13.088075+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.01208","created_at":"2026-07-05T10:05:13.088075+00:00"},{"alias_kind":"pith_short_12","alias_value":"3FCHIDRQ7EGD","created_at":"2026-07-05T10:05:13.088075+00:00"},{"alias_kind":"pith_short_16","alias_value":"3FCHIDRQ7EGDRCST","created_at":"2026-07-05T10:05:13.088075+00:00"},{"alias_kind":"pith_short_8","alias_value":"3FCHIDRQ","created_at":"2026-07-05T10:05:13.088075+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.10641","citing_title":"Spelling-out is not Straightforward: LLMs' Capability of Tokenization from Token to Characters","ref_index":26,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3FCHIDRQ7EGDRCSTUYJNMA52NG","json":"https://pith.science/pith/3FCHIDRQ7EGDRCSTUYJNMA52NG.json","graph_json":"https://pith.science/api/pith-number/3FCHIDRQ7EGDRCSTUYJNMA52NG/graph.json","events_json":"https://pith.science/api/pith-number/3FCHIDRQ7EGDRCSTUYJNMA52NG/events.json","paper":"https://pith.science/paper/3FCHIDRQ"},"agent_actions":{"view_html":"https://pith.science/pith/3FCHIDRQ7EGDRCSTUYJNMA52NG","download_json":"https://pith.science/pith/3FCHIDRQ7EGDRCSTUYJNMA52NG.json","view_paper":"https://pith.science/paper/3FCHIDRQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.01208&json=true","fetch_graph":"https://pith.science/api/pith-number/3FCHIDRQ7EGDRCSTUYJNMA52NG/graph.json","fetch_events":"https://pith.science/api/pith-number/3FCHIDRQ7EGDRCSTUYJNMA52NG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3FCHIDRQ7EGDRCSTUYJNMA52NG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3FCHIDRQ7EGDRCSTUYJNMA52NG/action/storage_attestation","attest_author":"https://pith.science/pith/3FCHIDRQ7EGDRCSTUYJNMA52NG/action/author_attestation","sign_citation":"https://pith.science/pith/3FCHIDRQ7EGDRCSTUYJNMA52NG/action/citation_signature","submit_replication":"https://pith.science/pith/3FCHIDRQ7EGDRCSTUYJNMA52NG/action/replication_record"}},"created_at":"2026-07-05T10:05:13.088075+00:00","updated_at":"2026-07-05T10:05:13.088075+00:00"}