{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JAXERQ6OW22WZNMHCVWBNQS5J2","short_pith_number":"pith:JAXERQ6O","schema_version":"1.0","canonical_sha256":"482e48c3ceb6b56cb587156c16c25d4e800b515ccd868acba341f8c51a6fe2e2","source":{"kind":"arxiv","id":"2505.23977","version":1},"attestation_state":"computed","paper":{"title":"VisualSphinx: Large-Scale Synthetic Vision Logic Puzzles for RL","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Bhaskar Ramasubramanian, Bill Yuchen Lin, Fengqing Jiang, Luyao Niu, Radha Poovendran, Yichen Feng, Yuetai Li, Zhangchen Xu","submitted_at":"2025-05-29T20:08:36Z","abstract_excerpt":"Vision language models (VLMs) are expected to perform effective multimodal reasoning and make logically coherent decisions, which is critical to tasks such as diagram understanding and spatial problem solving. However, current VLM reasoning lacks large-scale and well-structured training datasets. To bridge this gap, we propose VisualSphinx, a first-of-its-kind large-scale synthetic visual logical reasoning training data. To tackle the challenge of image synthesis with grounding answers, we propose a rule-to-image synthesis pipeline, which extracts and expands puzzle rules from seed questions a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.23977","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-29T20:08:36Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"93f5944fc55589cbf0f0daa326003d9ad5693b141bc605f61e33b18355204f9e","abstract_canon_sha256":"6df99dc8a47854e061e032d7915df2c638407e987263a8ada90e9d13af1c60dc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:39.405267Z","signature_b64":"Yw4lIAH9rfkXj9h+BENjQynfNNIz93LmxvDP7bAyL2RUQ4X95ljWROEhu9UupizDcft4KI9TjTb4YVGhK9LdBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"482e48c3ceb6b56cb587156c16c25d4e800b515ccd868acba341f8c51a6fe2e2","last_reissued_at":"2026-07-05T11:12:39.404753Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:39.404753Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VisualSphinx: Large-Scale Synthetic Vision Logic Puzzles for RL","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Bhaskar Ramasubramanian, Bill Yuchen Lin, Fengqing Jiang, Luyao Niu, Radha Poovendran, Yichen Feng, Yuetai Li, Zhangchen Xu","submitted_at":"2025-05-29T20:08:36Z","abstract_excerpt":"Vision language models (VLMs) are expected to perform effective multimodal reasoning and make logically coherent decisions, which is critical to tasks such as diagram understanding and spatial problem solving. However, current VLM reasoning lacks large-scale and well-structured training datasets. To bridge this gap, we propose VisualSphinx, a first-of-its-kind large-scale synthetic visual logical reasoning training data. To tackle the challenge of image synthesis with grounding answers, we propose a rule-to-image synthesis pipeline, which extracts and expands puzzle rules from seed questions a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23977","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.23977/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.23977","created_at":"2026-07-05T11:12:39.404819+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.23977v1","created_at":"2026-07-05T11:12:39.404819+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23977","created_at":"2026-07-05T11:12:39.404819+00:00"},{"alias_kind":"pith_short_12","alias_value":"JAXERQ6OW22W","created_at":"2026-07-05T11:12:39.404819+00:00"},{"alias_kind":"pith_short_16","alias_value":"JAXERQ6OW22WZNMH","created_at":"2026-07-05T11:12:39.404819+00:00"},{"alias_kind":"pith_short_8","alias_value":"JAXERQ6O","created_at":"2026-07-05T11:12:39.404819+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18216","citing_title":"Zone of Proximal Policy Optimization: Teacher in Prompts, Not Gradients","ref_index":127,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09567","citing_title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","ref_index":186,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JAXERQ6OW22WZNMHCVWBNQS5J2","json":"https://pith.science/pith/JAXERQ6OW22WZNMHCVWBNQS5J2.json","graph_json":"https://pith.science/api/pith-number/JAXERQ6OW22WZNMHCVWBNQS5J2/graph.json","events_json":"https://pith.science/api/pith-number/JAXERQ6OW22WZNMHCVWBNQS5J2/events.json","paper":"https://pith.science/paper/JAXERQ6O"},"agent_actions":{"view_html":"https://pith.science/pith/JAXERQ6OW22WZNMHCVWBNQS5J2","download_json":"https://pith.science/pith/JAXERQ6OW22WZNMHCVWBNQS5J2.json","view_paper":"https://pith.science/paper/JAXERQ6O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.23977&json=true","fetch_graph":"https://pith.science/api/pith-number/JAXERQ6OW22WZNMHCVWBNQS5J2/graph.json","fetch_events":"https://pith.science/api/pith-number/JAXERQ6OW22WZNMHCVWBNQS5J2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JAXERQ6OW22WZNMHCVWBNQS5J2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JAXERQ6OW22WZNMHCVWBNQS5J2/action/storage_attestation","attest_author":"https://pith.science/pith/JAXERQ6OW22WZNMHCVWBNQS5J2/action/author_attestation","sign_citation":"https://pith.science/pith/JAXERQ6OW22WZNMHCVWBNQS5J2/action/citation_signature","submit_replication":"https://pith.science/pith/JAXERQ6OW22WZNMHCVWBNQS5J2/action/replication_record"}},"created_at":"2026-07-05T11:12:39.404819+00:00","updated_at":"2026-07-05T11:12:39.404819+00:00"}