{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4RYRROCNR2IZXJEARFE56PXBXT","short_pith_number":"pith:4RYRROCN","schema_version":"1.0","canonical_sha256":"e47118b84d8e919ba4808949df3ee1bccd634a02cae190cc5e2778aeca95d77d","source":{"kind":"arxiv","id":"2505.23295","version":1},"attestation_state":"computed","paper":{"title":"How Does Response Length Affect Long-Form Factuality","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bryan Hooi, James Xu Zhao, Jimmy Z.J. Liu, See-kiong Ng","submitted_at":"2025-05-29T09:47:56Z","abstract_excerpt":"Large language models (LLMs) are widely used for long-form text generation. However, factual errors in the responses would undermine their reliability. Despite growing attention to LLM factuality, the effect of response length on factuality remains underexplored. In this work, we systematically investigate this relationship by first introducing an automatic and bi-level long-form factuality evaluation framework, which achieves high agreement with human annotations while being cost-effective. Using this framework, we conduct controlled experiments and find that longer responses exhibit lower fa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.23295","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-29T09:47:56Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"fca797f487f3d44cc1cc602a3bfeda4dc8dc45c73aa53ff0b3963f8f9df1d79b","abstract_canon_sha256":"f4741c482e8f61d10989aec4111e7f05703e6e1f75b6b71544eb03e56101332d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:59.034813Z","signature_b64":"ZkdtbAIcZHphHVVpgwEo/9iZg/DVQqRISC4OpxaZ290w/k2vE74MEIIfG/TIkukyYnMvLw5qnWU/7pIi1PRrAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e47118b84d8e919ba4808949df3ee1bccd634a02cae190cc5e2778aeca95d77d","last_reissued_at":"2026-07-05T11:11:59.034254Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:59.034254Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Does Response Length Affect Long-Form Factuality","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bryan Hooi, James Xu Zhao, Jimmy Z.J. Liu, See-kiong Ng","submitted_at":"2025-05-29T09:47:56Z","abstract_excerpt":"Large language models (LLMs) are widely used for long-form text generation. However, factual errors in the responses would undermine their reliability. Despite growing attention to LLM factuality, the effect of response length on factuality remains underexplored. In this work, we systematically investigate this relationship by first introducing an automatic and bi-level long-form factuality evaluation framework, which achieves high agreement with human annotations while being cost-effective. Using this framework, we conduct controlled experiments and find that longer responses exhibit lower fa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23295","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.23295/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.23295","created_at":"2026-07-05T11:11:59.034320+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.23295v1","created_at":"2026-07-05T11:11:59.034320+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23295","created_at":"2026-07-05T11:11:59.034320+00:00"},{"alias_kind":"pith_short_12","alias_value":"4RYRROCNR2IZ","created_at":"2026-07-05T11:11:59.034320+00:00"},{"alias_kind":"pith_short_16","alias_value":"4RYRROCNR2IZXJEA","created_at":"2026-07-05T11:11:59.034320+00:00"},{"alias_kind":"pith_short_8","alias_value":"4RYRROCN","created_at":"2026-07-05T11:11:59.034320+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4RYRROCNR2IZXJEARFE56PXBXT","json":"https://pith.science/pith/4RYRROCNR2IZXJEARFE56PXBXT.json","graph_json":"https://pith.science/api/pith-number/4RYRROCNR2IZXJEARFE56PXBXT/graph.json","events_json":"https://pith.science/api/pith-number/4RYRROCNR2IZXJEARFE56PXBXT/events.json","paper":"https://pith.science/paper/4RYRROCN"},"agent_actions":{"view_html":"https://pith.science/pith/4RYRROCNR2IZXJEARFE56PXBXT","download_json":"https://pith.science/pith/4RYRROCNR2IZXJEARFE56PXBXT.json","view_paper":"https://pith.science/paper/4RYRROCN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.23295&json=true","fetch_graph":"https://pith.science/api/pith-number/4RYRROCNR2IZXJEARFE56PXBXT/graph.json","fetch_events":"https://pith.science/api/pith-number/4RYRROCNR2IZXJEARFE56PXBXT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4RYRROCNR2IZXJEARFE56PXBXT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4RYRROCNR2IZXJEARFE56PXBXT/action/storage_attestation","attest_author":"https://pith.science/pith/4RYRROCNR2IZXJEARFE56PXBXT/action/author_attestation","sign_citation":"https://pith.science/pith/4RYRROCNR2IZXJEARFE56PXBXT/action/citation_signature","submit_replication":"https://pith.science/pith/4RYRROCNR2IZXJEARFE56PXBXT/action/replication_record"}},"created_at":"2026-07-05T11:11:59.034320+00:00","updated_at":"2026-07-05T11:11:59.034320+00:00"}