{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:T2C5VOHK7P4RW5U2YW7BVJJA3O","short_pith_number":"pith:T2C5VOHK","schema_version":"1.0","canonical_sha256":"9e85dab8eafbf91b769ac5be1aa520dba3259eacfcd8678472386b6bdf4a0ca6","source":{"kind":"arxiv","id":"2502.20758","version":3},"attestation_state":"computed","paper":{"title":"Collective Reasoning Among LLMs: A Framework for Answer Validation Without Ground Truth","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"stat.AP","authors_text":"Alireza Amiri-Margavi, Alireza Shafiee Fard, Amin Gholami Davodi, Mahdi Jafari, Seyed Pouyan Mousavi Davoudi","submitted_at":"2025-02-28T06:20:52Z","abstract_excerpt":"We introduce a new approach in which several advanced large language models-specifically GPT-4-0125-preview, Meta-LLAMA-3-70B-Instruct, Claude-3-Opus, and Gemini-1.5-Flash-collaborate to both produce and answer intricate, doctoral-level probability problems without relying on any single \"correct\" reference. Rather than depending on an established ground truth, our investigation focuses on how agreement among diverse models can signal the reliability of their outputs and, by extension, reflect the overall quality of the generated questions. To measure this inter-model alignment, we apply a suit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.20758","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"stat.AP","submitted_at":"2025-02-28T06:20:52Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"301bcae4873eae98b0ad8aea1646bd6d9fa572dd603108ba8431347117b6654f","abstract_canon_sha256":"c9c37c206f578bf70a25f8c8dfd6fccfa824e4cf8199ea5c51180fb9b3dcfe6f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:50:59.030009Z","signature_b64":"443vImrxjj7+Xpq1iQT9vaj5j+KGjhpDzIKxcBWfD/D3CpNPv4UDarEYKsEu8zZ5sVaFkEBji91DloQDTIyIDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9e85dab8eafbf91b769ac5be1aa520dba3259eacfcd8678472386b6bdf4a0ca6","last_reissued_at":"2026-07-05T11:50:59.029526Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:50:59.029526Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Collective Reasoning Among LLMs: A Framework for Answer Validation Without Ground Truth","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"stat.AP","authors_text":"Alireza Amiri-Margavi, Alireza Shafiee Fard, Amin Gholami Davodi, Mahdi Jafari, Seyed Pouyan Mousavi Davoudi","submitted_at":"2025-02-28T06:20:52Z","abstract_excerpt":"We introduce a new approach in which several advanced large language models-specifically GPT-4-0125-preview, Meta-LLAMA-3-70B-Instruct, Claude-3-Opus, and Gemini-1.5-Flash-collaborate to both produce and answer intricate, doctoral-level probability problems without relying on any single \"correct\" reference. Rather than depending on an established ground truth, our investigation focuses on how agreement among diverse models can signal the reliability of their outputs and, by extension, reflect the overall quality of the generated questions. To measure this inter-model alignment, we apply a suit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.20758","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.20758/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.20758","created_at":"2026-07-05T11:50:59.029585+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.20758v3","created_at":"2026-07-05T11:50:59.029585+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.20758","created_at":"2026-07-05T11:50:59.029585+00:00"},{"alias_kind":"pith_short_12","alias_value":"T2C5VOHK7P4R","created_at":"2026-07-05T11:50:59.029585+00:00"},{"alias_kind":"pith_short_16","alias_value":"T2C5VOHK7P4RW5U2","created_at":"2026-07-05T11:50:59.029585+00:00"},{"alias_kind":"pith_short_8","alias_value":"T2C5VOHK","created_at":"2026-07-05T11:50:59.029585+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01456","citing_title":"Truthful AI Advisors: A Pre-Specified Benchmark for Large Language Model Honesty Under Preference Misalignment","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T2C5VOHK7P4RW5U2YW7BVJJA3O","json":"https://pith.science/pith/T2C5VOHK7P4RW5U2YW7BVJJA3O.json","graph_json":"https://pith.science/api/pith-number/T2C5VOHK7P4RW5U2YW7BVJJA3O/graph.json","events_json":"https://pith.science/api/pith-number/T2C5VOHK7P4RW5U2YW7BVJJA3O/events.json","paper":"https://pith.science/paper/T2C5VOHK"},"agent_actions":{"view_html":"https://pith.science/pith/T2C5VOHK7P4RW5U2YW7BVJJA3O","download_json":"https://pith.science/pith/T2C5VOHK7P4RW5U2YW7BVJJA3O.json","view_paper":"https://pith.science/paper/T2C5VOHK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.20758&json=true","fetch_graph":"https://pith.science/api/pith-number/T2C5VOHK7P4RW5U2YW7BVJJA3O/graph.json","fetch_events":"https://pith.science/api/pith-number/T2C5VOHK7P4RW5U2YW7BVJJA3O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T2C5VOHK7P4RW5U2YW7BVJJA3O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T2C5VOHK7P4RW5U2YW7BVJJA3O/action/storage_attestation","attest_author":"https://pith.science/pith/T2C5VOHK7P4RW5U2YW7BVJJA3O/action/author_attestation","sign_citation":"https://pith.science/pith/T2C5VOHK7P4RW5U2YW7BVJJA3O/action/citation_signature","submit_replication":"https://pith.science/pith/T2C5VOHK7P4RW5U2YW7BVJJA3O/action/replication_record"}},"created_at":"2026-07-05T11:50:59.029585+00:00","updated_at":"2026-07-05T11:50:59.029585+00:00"}