{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OIBYEV2KIVGQKHCTSHEQTLJ4EK","short_pith_number":"pith:OIBYEV2K","schema_version":"1.0","canonical_sha256":"720382574a454d051c5391c909ad3c22b24781a96f2d2ed75f5d2d64e5244e1c","source":{"kind":"arxiv","id":"2307.16877","version":2},"attestation_state":"computed","paper":{"title":"Evaluating Correctness and Faithfulness of Instruction-Following Models for Question Answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Nicholas Meade, Parishad BehnamGhader, Siva Reddy, Vaibhav Adlakha, Xing Han Lu","submitted_at":"2023-07-31T17:41:00Z","abstract_excerpt":"Retriever-augmented instruction-following models are attractive alternatives to fine-tuned approaches for information-seeking tasks such as question answering (QA). By simply prepending retrieved documents in its input along with an instruction, these models can be adapted to various information domains and tasks without additional fine-tuning. While the model responses tend to be natural and fluent, the additional verbosity makes traditional QA evaluation metrics such as exact match (EM) and F1 unreliable for accurately quantifying model performance.\n  In this work, we investigate the perform"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.16877","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-07-31T17:41:00Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"82d1271cc9300dcbb2b323a44389f4f0920d6ffad45e71956fdd17e8dfbb0370","abstract_canon_sha256":"c40e9c31def3490f50a9b14fe572baf60fa1ccc729f156b663b6059cb2fd607d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:08:50.307083Z","signature_b64":"03eq69JE9QEERJoUMchQ/bapXk383NCCxFOLzjaUIyz47EXR26gA1BB5tUQ8DuV4Ku76+kDXFThHBF1PwaYwCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"720382574a454d051c5391c909ad3c22b24781a96f2d2ed75f5d2d64e5244e1c","last_reissued_at":"2026-07-05T08:08:50.306624Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:08:50.306624Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Correctness and Faithfulness of Instruction-Following Models for Question Answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Nicholas Meade, Parishad BehnamGhader, Siva Reddy, Vaibhav Adlakha, Xing Han Lu","submitted_at":"2023-07-31T17:41:00Z","abstract_excerpt":"Retriever-augmented instruction-following models are attractive alternatives to fine-tuned approaches for information-seeking tasks such as question answering (QA). By simply prepending retrieved documents in its input along with an instruction, these models can be adapted to various information domains and tasks without additional fine-tuning. While the model responses tend to be natural and fluent, the additional verbosity makes traditional QA evaluation metrics such as exact match (EM) and F1 unreliable for accurately quantifying model performance.\n  In this work, we investigate the perform"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.16877","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.16877/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.16877","created_at":"2026-07-05T08:08:50.306685+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.16877v2","created_at":"2026-07-05T08:08:50.306685+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.16877","created_at":"2026-07-05T08:08:50.306685+00:00"},{"alias_kind":"pith_short_12","alias_value":"OIBYEV2KIVGQ","created_at":"2026-07-05T08:08:50.306685+00:00"},{"alias_kind":"pith_short_16","alias_value":"OIBYEV2KIVGQKHCT","created_at":"2026-07-05T08:08:50.306685+00:00"},{"alias_kind":"pith_short_8","alias_value":"OIBYEV2K","created_at":"2026-07-05T08:08:50.306685+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28737","citing_title":"5ting at SemEval-2026 Task 8: Strong End-to-End Multi-Turn RAG via LLM-Based Reranking and Faithfulness Control","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2503.22693","citing_title":"Bridging Language Models and Financial Analysis","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2311.05232","citing_title":"A Survey on Hallucination in Large Language Models: Principles, Taxonomy, Challenges, and Open Questions","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OIBYEV2KIVGQKHCTSHEQTLJ4EK","json":"https://pith.science/pith/OIBYEV2KIVGQKHCTSHEQTLJ4EK.json","graph_json":"https://pith.science/api/pith-number/OIBYEV2KIVGQKHCTSHEQTLJ4EK/graph.json","events_json":"https://pith.science/api/pith-number/OIBYEV2KIVGQKHCTSHEQTLJ4EK/events.json","paper":"https://pith.science/paper/OIBYEV2K"},"agent_actions":{"view_html":"https://pith.science/pith/OIBYEV2KIVGQKHCTSHEQTLJ4EK","download_json":"https://pith.science/pith/OIBYEV2KIVGQKHCTSHEQTLJ4EK.json","view_paper":"https://pith.science/paper/OIBYEV2K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.16877&json=true","fetch_graph":"https://pith.science/api/pith-number/OIBYEV2KIVGQKHCTSHEQTLJ4EK/graph.json","fetch_events":"https://pith.science/api/pith-number/OIBYEV2KIVGQKHCTSHEQTLJ4EK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OIBYEV2KIVGQKHCTSHEQTLJ4EK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OIBYEV2KIVGQKHCTSHEQTLJ4EK/action/storage_attestation","attest_author":"https://pith.science/pith/OIBYEV2KIVGQKHCTSHEQTLJ4EK/action/author_attestation","sign_citation":"https://pith.science/pith/OIBYEV2KIVGQKHCTSHEQTLJ4EK/action/citation_signature","submit_replication":"https://pith.science/pith/OIBYEV2KIVGQKHCTSHEQTLJ4EK/action/replication_record"}},"created_at":"2026-07-05T08:08:50.306685+00:00","updated_at":"2026-07-05T08:08:50.306685+00:00"}