{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NFG7CKDSIAVJ3ROYCE74JB5V23","short_pith_number":"pith:NFG7CKDS","schema_version":"1.0","canonical_sha256":"694df12872402a9dc5d8113fc487b5d6e455ab4bd4906c59c25f21c93c124413","source":{"kind":"arxiv","id":"2501.00332","version":1},"attestation_state":"computed","paper":{"title":"MAIN-RAG: Multi-Agent Filtering Retrieval-Augmented Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Chia-Yuan Chang, Chin-Chia Michael Yeh, Guanchu Wang, Mahashweta Das, Menghai Pan, Mingzhi Hu, Na Zou, Vineeth Rakesh, Yan Zheng, Zhichao Xu, Zhimeng Jiang","submitted_at":"2024-12-31T08:07:26Z","abstract_excerpt":"Large Language Models (LLMs) are becoming essential tools for various natural language processing tasks but often suffer from generating outdated or incorrect information. Retrieval-Augmented Generation (RAG) addresses this issue by incorporating external, real-time information retrieval to ground LLM responses. However, the existing RAG systems frequently struggle with the quality of retrieval documents, as irrelevant or noisy documents degrade performance, increase computational overhead, and undermine response reliability. To tackle this problem, we propose Multi-Agent Filtering Retrieval-A"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.00332","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-12-31T08:07:26Z","cross_cats_sorted":["cs.IR"],"title_canon_sha256":"16d3196b74e42c06b28d8e862763841878d43959a4bdf7dd405356a75606551a","abstract_canon_sha256":"07ad129eabdb6fc6d17345e3f0c5df3ebf9dae2b6094cb4fde4e2211abc0472c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:55:43.104295Z","signature_b64":"q68D1HyLPpU8iaIB/2oZ2lrWjk5u5JaOTqN18B1V7vn7SzyJkVzWwhuje8qXOdVLx+nVYlvQLh4/lPEDh9aQDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"694df12872402a9dc5d8113fc487b5d6e455ab4bd4906c59c25f21c93c124413","last_reissued_at":"2026-07-05T09:55:43.103900Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:55:43.103900Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MAIN-RAG: Multi-Agent Filtering Retrieval-Augmented Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Chia-Yuan Chang, Chin-Chia Michael Yeh, Guanchu Wang, Mahashweta Das, Menghai Pan, Mingzhi Hu, Na Zou, Vineeth Rakesh, Yan Zheng, Zhichao Xu, Zhimeng Jiang","submitted_at":"2024-12-31T08:07:26Z","abstract_excerpt":"Large Language Models (LLMs) are becoming essential tools for various natural language processing tasks but often suffer from generating outdated or incorrect information. Retrieval-Augmented Generation (RAG) addresses this issue by incorporating external, real-time information retrieval to ground LLM responses. However, the existing RAG systems frequently struggle with the quality of retrieval documents, as irrelevant or noisy documents degrade performance, increase computational overhead, and undermine response reliability. To tackle this problem, we propose Multi-Agent Filtering Retrieval-A"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.00332","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.00332/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.00332","created_at":"2026-07-05T09:55:43.103956+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.00332v1","created_at":"2026-07-05T09:55:43.103956+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.00332","created_at":"2026-07-05T09:55:43.103956+00:00"},{"alias_kind":"pith_short_12","alias_value":"NFG7CKDSIAVJ","created_at":"2026-07-05T09:55:43.103956+00:00"},{"alias_kind":"pith_short_16","alias_value":"NFG7CKDSIAVJ3ROY","created_at":"2026-07-05T09:55:43.103956+00:00"},{"alias_kind":"pith_short_8","alias_value":"NFG7CKDS","created_at":"2026-07-05T09:55:43.103956+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.01248","citing_title":"$S^3$-R1: Learning to Retrieve and Answer Step-by-Step with Synthetic Data","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2603.22453","citing_title":"XNote: Benchmarking Automated Community Notes Generation for Image-based Contextual Deception","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2511.01008","citing_title":"MARS-SQL: A multi-agent reinforcement learning framework for Text-to-SQL","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2511.20857","citing_title":"Evo-Memory: Benchmarking LLM Agent Test-time Learning with Self-Evolving Memory","ref_index":197,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01248","citing_title":"$S^3$-R1: Learning to Retrieve and Answer Step-by-Step with Synthetic Data","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NFG7CKDSIAVJ3ROYCE74JB5V23","json":"https://pith.science/pith/NFG7CKDSIAVJ3ROYCE74JB5V23.json","graph_json":"https://pith.science/api/pith-number/NFG7CKDSIAVJ3ROYCE74JB5V23/graph.json","events_json":"https://pith.science/api/pith-number/NFG7CKDSIAVJ3ROYCE74JB5V23/events.json","paper":"https://pith.science/paper/NFG7CKDS"},"agent_actions":{"view_html":"https://pith.science/pith/NFG7CKDSIAVJ3ROYCE74JB5V23","download_json":"https://pith.science/pith/NFG7CKDSIAVJ3ROYCE74JB5V23.json","view_paper":"https://pith.science/paper/NFG7CKDS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.00332&json=true","fetch_graph":"https://pith.science/api/pith-number/NFG7CKDSIAVJ3ROYCE74JB5V23/graph.json","fetch_events":"https://pith.science/api/pith-number/NFG7CKDSIAVJ3ROYCE74JB5V23/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NFG7CKDSIAVJ3ROYCE74JB5V23/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NFG7CKDSIAVJ3ROYCE74JB5V23/action/storage_attestation","attest_author":"https://pith.science/pith/NFG7CKDSIAVJ3ROYCE74JB5V23/action/author_attestation","sign_citation":"https://pith.science/pith/NFG7CKDSIAVJ3ROYCE74JB5V23/action/citation_signature","submit_replication":"https://pith.science/pith/NFG7CKDSIAVJ3ROYCE74JB5V23/action/replication_record"}},"created_at":"2026-07-05T09:55:43.103956+00:00","updated_at":"2026-07-05T09:55:43.103956+00:00"}