{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3F75H4CFI3KI63IZ3F7IBGKHDW","short_pith_number":"pith:3F75H4CF","schema_version":"1.0","canonical_sha256":"d97fd3f04546d48f6d19d97e8099471dbc21c5f9d8ffaae98f4e3bc53cd134b8","source":{"kind":"arxiv","id":"2406.12585","version":2},"attestation_state":"computed","paper":{"title":"Breaking the Ceiling of the LLM Community by Treating Token Generation as a Classification for Ensembling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chun-Chih Kuo, Yao-Ching Yu, Yu-Cheng Chang, Yueh-Se Li, Ziqi Ye","submitted_at":"2024-06-18T13:17:26Z","abstract_excerpt":"Ensembling multiple models has always been an effective approach to push the limits of existing performance and is widely used in classification tasks by simply averaging the classification probability vectors from multiple classifiers to achieve better accuracy. However, in the thriving open-source Large Language Model (LLM) community, ensembling methods are rare and typically limited to ensembling the full-text outputs of LLMs, such as selecting the best output using a ranker, which leads to underutilization of token-level probability information. In this paper, we treat the Generation of ea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.12585","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-18T13:17:26Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ae0ecbdf3482cb41dbf369af573db4de7002f82cc79489e0925922a483c364d7","abstract_canon_sha256":"f63318369dcc5264f44f6f59829b55aec26c7f1dfe2e2fe4d7326e753182a0db"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:13:04.163960Z","signature_b64":"uztRFi9NRpPid1DkKue7nsZR/LJRFtU+pet6DCBW7UR3Ala5J8lRqPxt1z3LFIXUh8cNMPiLUrDk1Pb8sJIZAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d97fd3f04546d48f6d19d97e8099471dbc21c5f9d8ffaae98f4e3bc53cd134b8","last_reissued_at":"2026-07-05T09:13:04.163485Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:13:04.163485Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Breaking the Ceiling of the LLM Community by Treating Token Generation as a Classification for Ensembling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chun-Chih Kuo, Yao-Ching Yu, Yu-Cheng Chang, Yueh-Se Li, Ziqi Ye","submitted_at":"2024-06-18T13:17:26Z","abstract_excerpt":"Ensembling multiple models has always been an effective approach to push the limits of existing performance and is widely used in classification tasks by simply averaging the classification probability vectors from multiple classifiers to achieve better accuracy. However, in the thriving open-source Large Language Model (LLM) community, ensembling methods are rare and typically limited to ensembling the full-text outputs of LLMs, such as selecting the best output using a ranker, which leads to underutilization of token-level probability information. In this paper, we treat the Generation of ea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.12585","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.12585/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.12585","created_at":"2026-07-05T09:13:04.163536+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.12585v2","created_at":"2026-07-05T09:13:04.163536+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.12585","created_at":"2026-07-05T09:13:04.163536+00:00"},{"alias_kind":"pith_short_12","alias_value":"3F75H4CFI3KI","created_at":"2026-07-05T09:13:04.163536+00:00"},{"alias_kind":"pith_short_16","alias_value":"3F75H4CFI3KI63IZ","created_at":"2026-07-05T09:13:04.163536+00:00"},{"alias_kind":"pith_short_8","alias_value":"3F75H4CF","created_at":"2026-07-05T09:13:04.163536+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.18036","citing_title":"Harnessing Multiple Large Language Models: A Survey on LLM Ensemble","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2512.23213","citing_title":"Scoring, Reasoning, and Selecting the Best! Ensembling Large Language Models via a Peer-Review Process","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3F75H4CFI3KI63IZ3F7IBGKHDW","json":"https://pith.science/pith/3F75H4CFI3KI63IZ3F7IBGKHDW.json","graph_json":"https://pith.science/api/pith-number/3F75H4CFI3KI63IZ3F7IBGKHDW/graph.json","events_json":"https://pith.science/api/pith-number/3F75H4CFI3KI63IZ3F7IBGKHDW/events.json","paper":"https://pith.science/paper/3F75H4CF"},"agent_actions":{"view_html":"https://pith.science/pith/3F75H4CFI3KI63IZ3F7IBGKHDW","download_json":"https://pith.science/pith/3F75H4CFI3KI63IZ3F7IBGKHDW.json","view_paper":"https://pith.science/paper/3F75H4CF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.12585&json=true","fetch_graph":"https://pith.science/api/pith-number/3F75H4CFI3KI63IZ3F7IBGKHDW/graph.json","fetch_events":"https://pith.science/api/pith-number/3F75H4CFI3KI63IZ3F7IBGKHDW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3F75H4CFI3KI63IZ3F7IBGKHDW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3F75H4CFI3KI63IZ3F7IBGKHDW/action/storage_attestation","attest_author":"https://pith.science/pith/3F75H4CFI3KI63IZ3F7IBGKHDW/action/author_attestation","sign_citation":"https://pith.science/pith/3F75H4CFI3KI63IZ3F7IBGKHDW/action/citation_signature","submit_replication":"https://pith.science/pith/3F75H4CFI3KI63IZ3F7IBGKHDW/action/replication_record"}},"created_at":"2026-07-05T09:13:04.163536+00:00","updated_at":"2026-07-05T09:13:04.163536+00:00"}