{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PRFI2SLJP44RFTVGHKASUUWCTS","short_pith_number":"pith:PRFI2SLJ","schema_version":"1.0","canonical_sha256":"7c4a8d49697f3912cea63a812a52c29cbde4cf5c1ca0c1dabdf9801e63245685","source":{"kind":"arxiv","id":"2503.04104","version":2},"attestation_state":"computed","paper":{"title":"LLMs Can Generate a Better Answer by Aggregating Their Own Responses","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chen Liang, Haoyu Wang, Tianyi Liu, Tuo Zhao, Weizhu Chen, Xinyu Feng, Yuheng Cai, Zichong Li, Zixuan Zhang","submitted_at":"2025-03-06T05:25:43Z","abstract_excerpt":"Large Language Models (LLMs) have shown remarkable capabilities across tasks, yet they often require additional prompting techniques when facing complex problems. While approaches like self-correction and response selection have emerged as popular solutions, recent studies have shown these methods perform poorly when relying on the LLM itself to provide feedback or selection criteria. We argue this limitation stems from the fact that common LLM post-training procedures lack explicit supervision for discriminative judgment tasks. In this paper, we propose Generative Self-Aggregation (GSA), a no"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.04104","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-06T05:25:43Z","cross_cats_sorted":[],"title_canon_sha256":"f66e4e4153f6de705bdd9d291151984a8f7563ffb7a762b9fd9188fb7c7001e5","abstract_canon_sha256":"1b4cc83e77a0852bfb0b10e56cf31ca2b54c65a288255e665c204f481569d39a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:48:26.428689Z","signature_b64":"J4DmwPneztk6qnEHK/tN1SBglksliJ6uqewLJbKpRRdsay/b0ESBXcgjm2QciErt86ttAvXunAHVYHt3QT6vCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7c4a8d49697f3912cea63a812a52c29cbde4cf5c1ca0c1dabdf9801e63245685","last_reissued_at":"2026-07-05T10:48:26.428166Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:48:26.428166Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLMs Can Generate a Better Answer by Aggregating Their Own Responses","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chen Liang, Haoyu Wang, Tianyi Liu, Tuo Zhao, Weizhu Chen, Xinyu Feng, Yuheng Cai, Zichong Li, Zixuan Zhang","submitted_at":"2025-03-06T05:25:43Z","abstract_excerpt":"Large Language Models (LLMs) have shown remarkable capabilities across tasks, yet they often require additional prompting techniques when facing complex problems. While approaches like self-correction and response selection have emerged as popular solutions, recent studies have shown these methods perform poorly when relying on the LLM itself to provide feedback or selection criteria. We argue this limitation stems from the fact that common LLM post-training procedures lack explicit supervision for discriminative judgment tasks. In this paper, we propose Generative Self-Aggregation (GSA), a no"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.04104","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.04104/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.04104","created_at":"2026-07-05T10:48:26.428234+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.04104v2","created_at":"2026-07-05T10:48:26.428234+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.04104","created_at":"2026-07-05T10:48:26.428234+00:00"},{"alias_kind":"pith_short_12","alias_value":"PRFI2SLJP44R","created_at":"2026-07-05T10:48:26.428234+00:00"},{"alias_kind":"pith_short_16","alias_value":"PRFI2SLJP44RFTVG","created_at":"2026-07-05T10:48:26.428234+00:00"},{"alias_kind":"pith_short_8","alias_value":"PRFI2SLJ","created_at":"2026-07-05T10:48:26.428234+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28864","citing_title":"On Test-Time Scaling for Vision-Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28864","citing_title":"On Test-Time Scaling for Vision-Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05868","citing_title":"Understanding Performance Gap Between Parallel and Sequential Sampling in Large Reasoning Models","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PRFI2SLJP44RFTVGHKASUUWCTS","json":"https://pith.science/pith/PRFI2SLJP44RFTVGHKASUUWCTS.json","graph_json":"https://pith.science/api/pith-number/PRFI2SLJP44RFTVGHKASUUWCTS/graph.json","events_json":"https://pith.science/api/pith-number/PRFI2SLJP44RFTVGHKASUUWCTS/events.json","paper":"https://pith.science/paper/PRFI2SLJ"},"agent_actions":{"view_html":"https://pith.science/pith/PRFI2SLJP44RFTVGHKASUUWCTS","download_json":"https://pith.science/pith/PRFI2SLJP44RFTVGHKASUUWCTS.json","view_paper":"https://pith.science/paper/PRFI2SLJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.04104&json=true","fetch_graph":"https://pith.science/api/pith-number/PRFI2SLJP44RFTVGHKASUUWCTS/graph.json","fetch_events":"https://pith.science/api/pith-number/PRFI2SLJP44RFTVGHKASUUWCTS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PRFI2SLJP44RFTVGHKASUUWCTS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PRFI2SLJP44RFTVGHKASUUWCTS/action/storage_attestation","attest_author":"https://pith.science/pith/PRFI2SLJP44RFTVGHKASUUWCTS/action/author_attestation","sign_citation":"https://pith.science/pith/PRFI2SLJP44RFTVGHKASUUWCTS/action/citation_signature","submit_replication":"https://pith.science/pith/PRFI2SLJP44RFTVGHKASUUWCTS/action/replication_record"}},"created_at":"2026-07-05T10:48:26.428234+00:00","updated_at":"2026-07-05T10:48:26.428234+00:00"}