{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:QXSZ2OGQU5PA2IZNMKBXIZA4ME","short_pith_number":"pith:QXSZ2OGQ","schema_version":"1.0","canonical_sha256":"85e59d38d0a75e0d232d628374641c610c7f6c16a7ad562c4fdced575e38a422","source":{"kind":"arxiv","id":"2209.14610","version":3},"attestation_state":"computed","paper":{"title":"Dynamic Prompt Learning via Policy Gradient for Semi-structured Mathematical Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Ashwin Kalyan, Kai-Wei Chang, Liang Qiu, Pan Lu, Peter Clark, Song-Chun Zhu, Tanmay Rajpurohit, Ying Nian Wu","submitted_at":"2022-09-29T08:01:04Z","abstract_excerpt":"Mathematical reasoning, a core ability of human intelligence, presents unique challenges for machines in abstract thinking and logical reasoning. Recent large pre-trained language models such as GPT-3 have achieved remarkable progress on mathematical reasoning tasks written in text form, such as math word problems (MWP). However, it is unknown if the models can handle more complex problems that involve math reasoning over heterogeneous information, such as tabular data. To fill the gap, we present Tabular Math Word Problems (TabMWP), a new dataset containing 38,431 open-domain grade-level prob"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.14610","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-09-29T08:01:04Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV"],"title_canon_sha256":"c2cbafd31e1ed0527fe8971b3cc167c93b63aa1eb9155c90981e0e31db48cdf8","abstract_canon_sha256":"5b685c5f8f23bd41a4c9252e1783ff6bb18ae44223fd61114f4a0ee3771e5667"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:47:17.700302Z","signature_b64":"vF9MjaHE8WVl/AiKXP4uXJjqjbDBV1/K0UxUGhpFxtstHiepHylAl0LtIv1iZSQfzPWEW+3XQJNMQygpLrfUAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"85e59d38d0a75e0d232d628374641c610c7f6c16a7ad562c4fdced575e38a422","last_reissued_at":"2026-07-05T05:47:17.699787Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:47:17.699787Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dynamic Prompt Learning via Policy Gradient for Semi-structured Mathematical Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Ashwin Kalyan, Kai-Wei Chang, Liang Qiu, Pan Lu, Peter Clark, Song-Chun Zhu, Tanmay Rajpurohit, Ying Nian Wu","submitted_at":"2022-09-29T08:01:04Z","abstract_excerpt":"Mathematical reasoning, a core ability of human intelligence, presents unique challenges for machines in abstract thinking and logical reasoning. Recent large pre-trained language models such as GPT-3 have achieved remarkable progress on mathematical reasoning tasks written in text form, such as math word problems (MWP). However, it is unknown if the models can handle more complex problems that involve math reasoning over heterogeneous information, such as tabular data. To fill the gap, we present Tabular Math Word Problems (TabMWP), a new dataset containing 38,431 open-domain grade-level prob"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.14610","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.14610/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.14610","created_at":"2026-07-05T05:47:17.699848+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.14610v3","created_at":"2026-07-05T05:47:17.699848+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.14610","created_at":"2026-07-05T05:47:17.699848+00:00"},{"alias_kind":"pith_short_12","alias_value":"QXSZ2OGQU5PA","created_at":"2026-07-05T05:47:17.699848+00:00"},{"alias_kind":"pith_short_16","alias_value":"QXSZ2OGQU5PA2IZN","created_at":"2026-07-05T05:47:17.699848+00:00"},{"alias_kind":"pith_short_8","alias_value":"QXSZ2OGQ","created_at":"2026-07-05T05:47:17.699848+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":21,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2604.17433","citing_title":"Self-Consistency from Only Two Samples: CoT-PoT Ensembling for Efficient LLM Reasoning","ref_index":2,"is_internal_anchor":true},{"citing_arxiv_id":"2606.11537","citing_title":"MoCA-Agent: A Market-of-Claims Code Agent for Financial and Numerical Reasoning","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10298","citing_title":"From Context-Aware to Conflict-Aware: Generalizing Contrastive Decoding for Knowledge Conflict in LLMs","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03606","citing_title":"Testing LLM Arithmetic Reasoning Generalization with Automatic Numeric-Remapping Attacks","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00390","citing_title":"Zamba2-VL Technical Report","ref_index":123,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31958","citing_title":"Adapting Generalist Robot Policies with Semantic Reinforcement Learning","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01075","citing_title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01018","citing_title":"WildTableBench: Benchmarking Multimodal Foundation Models on Table Understanding In the Wild","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2504.09925","citing_title":"FLARE: Fully Integration of Vision-Language Representations for Deep Cross-Modal Understanding","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2510.09671","citing_title":"Table Question Answering in the Era of Large Language Models: A Comprehensive Survey of Tasks, Methods, and Evaluation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2407.03320","citing_title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2601.03682","citing_title":"From Implicit to Explicit: Token-Efficient Logical Supervision for Mathematical Reasoning in LLMs","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2503.07536","citing_title":"LMM-R1: Empowering 3B LMMs with Strong Reasoning Abilities Through Two-Stage Rule-Based RL","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2309.17421","citing_title":"The Dawn of LMMs: Preliminary Explorations with GPT-4V(ision)","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16821","citing_title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2302.00923","citing_title":"Multimodal Chain-of-Thought Reasoning in Language Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2211.12588","citing_title":"Program of Thoughts Prompting: Disentangling Computation from Reasoning for Numerical Reasoning Tasks","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08560","citing_title":"ZAYA1-VL-8B Technical Report","ref_index":163,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01018","citing_title":"WildTableBench: Benchmarking Multimodal Foundation Models on Table Understanding In the Wild","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20755","citing_title":"V-tableR1: Process-Supervised Multimodal Table Reasoning with Critic-Guided Policy Optimization","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05271","citing_title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","ref_index":166,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QXSZ2OGQU5PA2IZNMKBXIZA4ME","json":"https://pith.science/pith/QXSZ2OGQU5PA2IZNMKBXIZA4ME.json","graph_json":"https://pith.science/api/pith-number/QXSZ2OGQU5PA2IZNMKBXIZA4ME/graph.json","events_json":"https://pith.science/api/pith-number/QXSZ2OGQU5PA2IZNMKBXIZA4ME/events.json","paper":"https://pith.science/paper/QXSZ2OGQ"},"agent_actions":{"view_html":"https://pith.science/pith/QXSZ2OGQU5PA2IZNMKBXIZA4ME","download_json":"https://pith.science/pith/QXSZ2OGQU5PA2IZNMKBXIZA4ME.json","view_paper":"https://pith.science/paper/QXSZ2OGQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.14610&json=true","fetch_graph":"https://pith.science/api/pith-number/QXSZ2OGQU5PA2IZNMKBXIZA4ME/graph.json","fetch_events":"https://pith.science/api/pith-number/QXSZ2OGQU5PA2IZNMKBXIZA4ME/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QXSZ2OGQU5PA2IZNMKBXIZA4ME/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QXSZ2OGQU5PA2IZNMKBXIZA4ME/action/storage_attestation","attest_author":"https://pith.science/pith/QXSZ2OGQU5PA2IZNMKBXIZA4ME/action/author_attestation","sign_citation":"https://pith.science/pith/QXSZ2OGQU5PA2IZNMKBXIZA4ME/action/citation_signature","submit_replication":"https://pith.science/pith/QXSZ2OGQU5PA2IZNMKBXIZA4ME/action/replication_record"}},"created_at":"2026-07-05T05:47:17.699848+00:00","updated_at":"2026-07-05T05:47:17.699848+00:00"}