{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FYCRH6TANLBGSOOX3H6HI24MV2","short_pith_number":"pith:FYCRH6TA","schema_version":"1.0","canonical_sha256":"2e0513fa606ac26939d7d9fc746b8caeac19757f4b40e9636b30213f4149e0c8","source":{"kind":"arxiv","id":"2410.03524","version":2},"attestation_state":"computed","paper":{"title":"Steering Large Language Models between Code Execution and Textual Reasoning","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chi Wang, Chuchu Fan, Harsh Jhamtani, Srinagesh Sharma, Yongchao Chen","submitted_at":"2024-10-04T15:44:47Z","abstract_excerpt":"While a lot of recent research focuses on enhancing the textual reasoning capabilities of Large Language Models (LLMs) by optimizing the multi-agent framework or reasoning chains, several benchmark tasks can be solved with 100\\% success through direct coding, which is more scalable and avoids the computational overhead associated with textual iterating and searching. Textual reasoning has inherent limitations in solving tasks with challenges in math, logics, optimization, and searching, which is unlikely to be solved by simply scaling up the model and data size. The recently released OpenAI GP"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.03524","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-04T15:44:47Z","cross_cats_sorted":[],"title_canon_sha256":"432e658df1fa2c10d2dbbf6955a6b774191d7c52d485e4db56c623f516bf588e","abstract_canon_sha256":"20d82786d32aeb18c81dd76e481952df1b75aad8349d98cded6bb4373c7c90a4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:05.953625Z","signature_b64":"r5LqOZFmovr4pBiWjBhV4BNFBbXEKlaNGv/P2v8P73yXwb8Uo4mu8j+MqLI2GvWrXOYQRnYNoA8I0Sz8xNF0Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2e0513fa606ac26939d7d9fc746b8caeac19757f4b40e9636b30213f4149e0c8","last_reissued_at":"2026-07-05T10:22:05.953067Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:05.953067Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Steering Large Language Models between Code Execution and Textual Reasoning","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chi Wang, Chuchu Fan, Harsh Jhamtani, Srinagesh Sharma, Yongchao Chen","submitted_at":"2024-10-04T15:44:47Z","abstract_excerpt":"While a lot of recent research focuses on enhancing the textual reasoning capabilities of Large Language Models (LLMs) by optimizing the multi-agent framework or reasoning chains, several benchmark tasks can be solved with 100\\% success through direct coding, which is more scalable and avoids the computational overhead associated with textual iterating and searching. Textual reasoning has inherent limitations in solving tasks with challenges in math, logics, optimization, and searching, which is unlikely to be solved by simply scaling up the model and data size. The recently released OpenAI GP"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.03524","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.03524/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.03524","created_at":"2026-07-05T10:22:05.953130+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.03524v2","created_at":"2026-07-05T10:22:05.953130+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.03524","created_at":"2026-07-05T10:22:05.953130+00:00"},{"alias_kind":"pith_short_12","alias_value":"FYCRH6TANLBG","created_at":"2026-07-05T10:22:05.953130+00:00"},{"alias_kind":"pith_short_16","alias_value":"FYCRH6TANLBGSOOX","created_at":"2026-07-05T10:22:05.953130+00:00"},{"alias_kind":"pith_short_8","alias_value":"FYCRH6TA","created_at":"2026-07-05T10:22:05.953130+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.12813","citing_title":"REALISTA: Realistic Latent Adversarial Attacks that Elicit LLM Hallucinations","ref_index":193,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04443","citing_title":"DeonticBench: A Benchmark for Reasoning over Rules","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FYCRH6TANLBGSOOX3H6HI24MV2","json":"https://pith.science/pith/FYCRH6TANLBGSOOX3H6HI24MV2.json","graph_json":"https://pith.science/api/pith-number/FYCRH6TANLBGSOOX3H6HI24MV2/graph.json","events_json":"https://pith.science/api/pith-number/FYCRH6TANLBGSOOX3H6HI24MV2/events.json","paper":"https://pith.science/paper/FYCRH6TA"},"agent_actions":{"view_html":"https://pith.science/pith/FYCRH6TANLBGSOOX3H6HI24MV2","download_json":"https://pith.science/pith/FYCRH6TANLBGSOOX3H6HI24MV2.json","view_paper":"https://pith.science/paper/FYCRH6TA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.03524&json=true","fetch_graph":"https://pith.science/api/pith-number/FYCRH6TANLBGSOOX3H6HI24MV2/graph.json","fetch_events":"https://pith.science/api/pith-number/FYCRH6TANLBGSOOX3H6HI24MV2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FYCRH6TANLBGSOOX3H6HI24MV2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FYCRH6TANLBGSOOX3H6HI24MV2/action/storage_attestation","attest_author":"https://pith.science/pith/FYCRH6TANLBGSOOX3H6HI24MV2/action/author_attestation","sign_citation":"https://pith.science/pith/FYCRH6TANLBGSOOX3H6HI24MV2/action/citation_signature","submit_replication":"https://pith.science/pith/FYCRH6TANLBGSOOX3H6HI24MV2/action/replication_record"}},"created_at":"2026-07-05T10:22:05.953130+00:00","updated_at":"2026-07-05T10:22:05.953130+00:00"}