{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:J7HRO5PZJJIOJFQR6N3CSEZSGJ","short_pith_number":"pith:J7HRO5PZ","schema_version":"1.0","canonical_sha256":"4fcf1775f94a50e49611f3762913323243e459605919c305dfa9a050d27b2e1f","source":{"kind":"arxiv","id":"2505.00903","version":1},"attestation_state":"computed","paper":{"title":"NeMo-Inspector: A Visualization Tool for LLM Generation Analysis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Daria Gitman, Evelina Bakhturina, Igor Gitman","submitted_at":"2025-05-01T22:47:06Z","abstract_excerpt":"Adapting Large Language Models (LLMs) to novel tasks and enhancing their overall capabilities often requires large, high-quality training datasets. Synthetic data, generated at scale, serves a valuable alternative when real-world data is scarce or difficult to obtain. However, ensuring the quality of synthetic datasets is challenging, as developers must manually inspect and refine numerous samples to identify errors and areas for improvement. This process is time-consuming and requires specialized tools. We introduce NeMo-Inspector, an open-source tool designed to simplify the analysis of synt"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.00903","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-01T22:47:06Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"20b743a189eeb4d3a97ad771471b6853f06bd25b01982e86a80bea710749ba05","abstract_canon_sha256":"e33f39fd1e0f6ec2d6c27f3c3461f7edb53d7cb7152b2ee580a2a78591884d3f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:57:41.482998Z","signature_b64":"YEcwlqVxGRpjn2TvOS8l2foDvfDNl1D1bNceOJlsMFUYMPehGR5DJx4ZW4HhzVODmuCXlwUloCH7pOyVVGiBAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4fcf1775f94a50e49611f3762913323243e459605919c305dfa9a050d27b2e1f","last_reissued_at":"2026-07-05T10:57:41.482569Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:57:41.482569Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NeMo-Inspector: A Visualization Tool for LLM Generation Analysis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Daria Gitman, Evelina Bakhturina, Igor Gitman","submitted_at":"2025-05-01T22:47:06Z","abstract_excerpt":"Adapting Large Language Models (LLMs) to novel tasks and enhancing their overall capabilities often requires large, high-quality training datasets. Synthetic data, generated at scale, serves a valuable alternative when real-world data is scarce or difficult to obtain. However, ensuring the quality of synthetic datasets is challenging, as developers must manually inspect and refine numerous samples to identify errors and areas for improvement. This process is time-consuming and requires specialized tools. We introduce NeMo-Inspector, an open-source tool designed to simplify the analysis of synt"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.00903","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.00903/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.00903","created_at":"2026-07-05T10:57:41.482631+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.00903v1","created_at":"2026-07-05T10:57:41.482631+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.00903","created_at":"2026-07-05T10:57:41.482631+00:00"},{"alias_kind":"pith_short_12","alias_value":"J7HRO5PZJJIO","created_at":"2026-07-05T10:57:41.482631+00:00"},{"alias_kind":"pith_short_16","alias_value":"J7HRO5PZJJIOJFQR","created_at":"2026-07-05T10:57:41.482631+00:00"},{"alias_kind":"pith_short_8","alias_value":"J7HRO5PZ","created_at":"2026-07-05T10:57:41.482631+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J7HRO5PZJJIOJFQR6N3CSEZSGJ","json":"https://pith.science/pith/J7HRO5PZJJIOJFQR6N3CSEZSGJ.json","graph_json":"https://pith.science/api/pith-number/J7HRO5PZJJIOJFQR6N3CSEZSGJ/graph.json","events_json":"https://pith.science/api/pith-number/J7HRO5PZJJIOJFQR6N3CSEZSGJ/events.json","paper":"https://pith.science/paper/J7HRO5PZ"},"agent_actions":{"view_html":"https://pith.science/pith/J7HRO5PZJJIOJFQR6N3CSEZSGJ","download_json":"https://pith.science/pith/J7HRO5PZJJIOJFQR6N3CSEZSGJ.json","view_paper":"https://pith.science/paper/J7HRO5PZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.00903&json=true","fetch_graph":"https://pith.science/api/pith-number/J7HRO5PZJJIOJFQR6N3CSEZSGJ/graph.json","fetch_events":"https://pith.science/api/pith-number/J7HRO5PZJJIOJFQR6N3CSEZSGJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J7HRO5PZJJIOJFQR6N3CSEZSGJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J7HRO5PZJJIOJFQR6N3CSEZSGJ/action/storage_attestation","attest_author":"https://pith.science/pith/J7HRO5PZJJIOJFQR6N3CSEZSGJ/action/author_attestation","sign_citation":"https://pith.science/pith/J7HRO5PZJJIOJFQR6N3CSEZSGJ/action/citation_signature","submit_replication":"https://pith.science/pith/J7HRO5PZJJIOJFQR6N3CSEZSGJ/action/replication_record"}},"created_at":"2026-07-05T10:57:41.482631+00:00","updated_at":"2026-07-05T10:57:41.482631+00:00"}