{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:R5GAB3QUN4ROPZELBQKGXCHRNY","short_pith_number":"pith:R5GAB3QU","schema_version":"1.0","canonical_sha256":"8f4c00ee146f22e7e48b0c146b88f16e3c8cc38fef6d043599182f03f7f061e4","source":{"kind":"arxiv","id":"2507.17797","version":1},"attestation_state":"computed","paper":{"title":"GenSelect: A Generative Approach to Best-of-N","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Aleksander Ficek, Igor Gitman, Ivan Moshkov, Ivan Sorokin, Shubham Toshniwal","submitted_at":"2025-07-23T15:22:51Z","abstract_excerpt":"Generative reward models with parallel sampling have enabled effective test-time scaling for reasoning tasks. Current approaches employ pointwise scoring of individual solutions or pairwise comparisons. However, pointwise methods underutilize LLMs' comparative abilities, while pairwise methods scale inefficiently with larger sampling budgets. We introduce GenSelect, where the LLM uses long reasoning to select the best solution among N candidates. This leverages LLMs' comparative strengths while scaling efficiently across parallel sampling budgets. For math reasoning, we demonstrate that reason"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.17797","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-07-23T15:22:51Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"70d2b6e25e8cd557396e5dada8c782fa75a8ce507243d492b147d558635b9f9c","abstract_canon_sha256":"5af1dd95aa8d59df0cf312ec16d56c82526b9274ad6245547d1e1ad2f6937056"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:42:35.447562Z","signature_b64":"EYZSI7fglkGg/YssIbTWZsuaQySqPl8z9CpN76/xlhUhSVae0H8zaROIgJisp4P+M+xDjFp1TyT694ET00/CDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8f4c00ee146f22e7e48b0c146b88f16e3c8cc38fef6d043599182f03f7f061e4","last_reissued_at":"2026-07-05T11:42:35.446966Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:42:35.446966Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GenSelect: A Generative Approach to Best-of-N","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Aleksander Ficek, Igor Gitman, Ivan Moshkov, Ivan Sorokin, Shubham Toshniwal","submitted_at":"2025-07-23T15:22:51Z","abstract_excerpt":"Generative reward models with parallel sampling have enabled effective test-time scaling for reasoning tasks. Current approaches employ pointwise scoring of individual solutions or pairwise comparisons. However, pointwise methods underutilize LLMs' comparative abilities, while pairwise methods scale inefficiently with larger sampling budgets. We introduce GenSelect, where the LLM uses long reasoning to select the best solution among N candidates. This leverages LLMs' comparative strengths while scaling efficiently across parallel sampling budgets. For math reasoning, we demonstrate that reason"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.17797","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.17797/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.17797","created_at":"2026-07-05T11:42:35.447026+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.17797v1","created_at":"2026-07-05T11:42:35.447026+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.17797","created_at":"2026-07-05T11:42:35.447026+00:00"},{"alias_kind":"pith_short_12","alias_value":"R5GAB3QUN4RO","created_at":"2026-07-05T11:42:35.447026+00:00"},{"alias_kind":"pith_short_16","alias_value":"R5GAB3QUN4ROPZEL","created_at":"2026-07-05T11:42:35.447026+00:00"},{"alias_kind":"pith_short_8","alias_value":"R5GAB3QU","created_at":"2026-07-05T11:42:35.447026+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.14232","citing_title":"Scaling Test-Time Compute to Achieve IOI Gold Medal with Open-Weight Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01194","citing_title":"VLA-ATTC: Adaptive Test-Time Compute for VLA Models with Relative Action Critic Model","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R5GAB3QUN4ROPZELBQKGXCHRNY","json":"https://pith.science/pith/R5GAB3QUN4ROPZELBQKGXCHRNY.json","graph_json":"https://pith.science/api/pith-number/R5GAB3QUN4ROPZELBQKGXCHRNY/graph.json","events_json":"https://pith.science/api/pith-number/R5GAB3QUN4ROPZELBQKGXCHRNY/events.json","paper":"https://pith.science/paper/R5GAB3QU"},"agent_actions":{"view_html":"https://pith.science/pith/R5GAB3QUN4ROPZELBQKGXCHRNY","download_json":"https://pith.science/pith/R5GAB3QUN4ROPZELBQKGXCHRNY.json","view_paper":"https://pith.science/paper/R5GAB3QU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.17797&json=true","fetch_graph":"https://pith.science/api/pith-number/R5GAB3QUN4ROPZELBQKGXCHRNY/graph.json","fetch_events":"https://pith.science/api/pith-number/R5GAB3QUN4ROPZELBQKGXCHRNY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R5GAB3QUN4ROPZELBQKGXCHRNY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R5GAB3QUN4ROPZELBQKGXCHRNY/action/storage_attestation","attest_author":"https://pith.science/pith/R5GAB3QUN4ROPZELBQKGXCHRNY/action/author_attestation","sign_citation":"https://pith.science/pith/R5GAB3QUN4ROPZELBQKGXCHRNY/action/citation_signature","submit_replication":"https://pith.science/pith/R5GAB3QUN4ROPZELBQKGXCHRNY/action/replication_record"}},"created_at":"2026-07-05T11:42:35.447026+00:00","updated_at":"2026-07-05T11:42:35.447026+00:00"}