{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:IXQT57G3SFDH2JH5E5A6TM6O2V","short_pith_number":"pith:IXQT57G3","schema_version":"1.0","canonical_sha256":"45e13efcdb91467d24fd2741e9b3ced5567d6d73a80aa003b3e385ddb115a30c","source":{"kind":"arxiv","id":"2105.03011","version":1},"attestation_state":"computed","paper":{"title":"A Dataset of Information-Seeking Questions and Answers Anchored in Research Papers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Arman Cohan, Iz Beltagy, Kyle Lo, Matt Gardner, Noah A. Smith, Pradeep Dasigi","submitted_at":"2021-05-07T00:12:34Z","abstract_excerpt":"Readers of academic research papers often read with the goal of answering specific questions. Question Answering systems that can answer those questions can make consumption of the content much more efficient. However, building such tools requires data that reflect the difficulty of the task arising from complex reasoning about claims made in multiple parts of a paper. In contrast, existing information-seeking question answering datasets usually contain questions about generic factoid-type information. We therefore present QASPER, a dataset of 5,049 questions over 1,585 Natural Language Proces"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2105.03011","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-05-07T00:12:34Z","cross_cats_sorted":[],"title_canon_sha256":"921b0ccf8c57d902e6cb056017a082a7a5ede7c0732ce59b4274f2bbcfc8b6b8","abstract_canon_sha256":"97694c62f2e74353de4d2aadcb9d27a7bdbac5d5b71e260c0292dcd8ef45102d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:38:20.581536Z","signature_b64":"7cg/TVaS34D75OGnqFHL1IOkQD91u9HuLtXA1rr4CcLqorudHZYSGHVwlpcz3lhM1z1+lsxw+sJ7uXlFNRUhAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"45e13efcdb91467d24fd2741e9b3ced5567d6d73a80aa003b3e385ddb115a30c","last_reissued_at":"2026-07-05T02:38:20.581009Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:38:20.581009Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Dataset of Information-Seeking Questions and Answers Anchored in Research Papers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Arman Cohan, Iz Beltagy, Kyle Lo, Matt Gardner, Noah A. Smith, Pradeep Dasigi","submitted_at":"2021-05-07T00:12:34Z","abstract_excerpt":"Readers of academic research papers often read with the goal of answering specific questions. Question Answering systems that can answer those questions can make consumption of the content much more efficient. However, building such tools requires data that reflect the difficulty of the task arising from complex reasoning about claims made in multiple parts of a paper. In contrast, existing information-seeking question answering datasets usually contain questions about generic factoid-type information. We therefore present QASPER, a dataset of 5,049 questions over 1,585 Natural Language Proces"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.03011","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.03011/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2105.03011","created_at":"2026-07-05T02:38:20.581076+00:00"},{"alias_kind":"arxiv_version","alias_value":"2105.03011v1","created_at":"2026-07-05T02:38:20.581076+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.03011","created_at":"2026-07-05T02:38:20.581076+00:00"},{"alias_kind":"pith_short_12","alias_value":"IXQT57G3SFDH","created_at":"2026-07-05T02:38:20.581076+00:00"},{"alias_kind":"pith_short_16","alias_value":"IXQT57G3SFDH2JH5","created_at":"2026-07-05T02:38:20.581076+00:00"},{"alias_kind":"pith_short_8","alias_value":"IXQT57G3","created_at":"2026-07-05T02:38:20.581076+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":23,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03979","citing_title":"Language Models Need Sleep: Learning to Self-Modify and Consolidate Memories","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20571","citing_title":"Less is More: Lightweight Prompt Compression for Question Answering Applications on Edge Devices","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09735","citing_title":"KV-RM: Regularizing KV-Cache Movement for Static-Graph LLM Serving","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29563","citing_title":"Coverage-Driven KV Cache Eviction for Efficient and Improved Inference of LLM","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00881","citing_title":"Chunking Methods on Retrieval-Augmented Generation - Effectiveness Evaluation Against Computational Cost and Limitations","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23704","citing_title":"PCB-QA: Evaluating LLMs over the First Printed Circuit Board Design Question-Answer Dataset","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2312.10997","citing_title":"Retrieval-Augmented Generation for Large Language Models: A Survey","ref_index":126,"is_internal_anchor":false},{"citing_arxiv_id":"2505.05772","citing_title":"Sparse Attention Remapping with Clustering for Efficient LLM Decoding on PIM","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2602.08686","citing_title":"CompilerKV: Risk-Adaptive KV Compression via Offline Experience Compilation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2506.10060","citing_title":"Textual Bayes: Quantifying Prompt Uncertainty in LLM-Based Systems","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2407.11550","citing_title":"Ada-KV: Optimizing KV Cache Eviction by Adaptive Budget Allocation for Efficient LLM Inference","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01203","citing_title":"Attention Sink Forges Native MoE in Attention Layers: Sink-Aware Training to Address Head Collapse","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12808","citing_title":"Neurodata Without Boredom: Benchmarking Agentic AI for Data Reuse","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12808","citing_title":"Neurodata Without Boredom: Benchmarking Agentic AI for Data Reuse","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14469","citing_title":"SnapKV: LLM Knows What You are Looking for Before Generation","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09735","citing_title":"KV-RM: Regularizing KV-Cache Movement for Static-Graph LLM Serving","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10171","citing_title":"When Reviews Disagree: Fine-Grained Contradiction Analysis in Scientific Peer Reviews","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10168","citing_title":"ASTRA-QA: A Benchmark for Abstract Question Answering over Documents","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25256","citing_title":"AutoResearchBench: Benchmarking AI Agents on Complex Scientific Literature Discovery","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07234","citing_title":"Reformulating KV Cache Eviction Problem for Long-Context LLM Inference","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17667","citing_title":"Peerispect: Claim Verification in Scientific Peer Reviews","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17112","citing_title":"Complementing Self-Consistency with Cross-Model Disagreement for Uncertainty Quantification","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17265","citing_title":"MemSearch-o1: Empowering Large Language Models with Reasoning-Aligned Memory Growth in Agentic Search","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IXQT57G3SFDH2JH5E5A6TM6O2V","json":"https://pith.science/pith/IXQT57G3SFDH2JH5E5A6TM6O2V.json","graph_json":"https://pith.science/api/pith-number/IXQT57G3SFDH2JH5E5A6TM6O2V/graph.json","events_json":"https://pith.science/api/pith-number/IXQT57G3SFDH2JH5E5A6TM6O2V/events.json","paper":"https://pith.science/paper/IXQT57G3"},"agent_actions":{"view_html":"https://pith.science/pith/IXQT57G3SFDH2JH5E5A6TM6O2V","download_json":"https://pith.science/pith/IXQT57G3SFDH2JH5E5A6TM6O2V.json","view_paper":"https://pith.science/paper/IXQT57G3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2105.03011&json=true","fetch_graph":"https://pith.science/api/pith-number/IXQT57G3SFDH2JH5E5A6TM6O2V/graph.json","fetch_events":"https://pith.science/api/pith-number/IXQT57G3SFDH2JH5E5A6TM6O2V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IXQT57G3SFDH2JH5E5A6TM6O2V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IXQT57G3SFDH2JH5E5A6TM6O2V/action/storage_attestation","attest_author":"https://pith.science/pith/IXQT57G3SFDH2JH5E5A6TM6O2V/action/author_attestation","sign_citation":"https://pith.science/pith/IXQT57G3SFDH2JH5E5A6TM6O2V/action/citation_signature","submit_replication":"https://pith.science/pith/IXQT57G3SFDH2JH5E5A6TM6O2V/action/replication_record"}},"created_at":"2026-07-05T02:38:20.581076+00:00","updated_at":"2026-07-05T02:38:20.581076+00:00"}