{"id":"W4413426951","doi":"10.21203/rs.3.rs-6959723/v1","title":"Retrieval-Augmented Generation for Natural Language Processing: A Survey","year":2025,"lang":"en","type":"preprint","venue":"Research Square","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University","funders":"","keywords":"Natural (archaeology); Computer science; Natural language processing; Information retrieval; Geography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002864687,0.001205179,0.002380642,0.002843552,0.0005833096,0.002508945,0.00345329,0.001523057,0.006974498],"category_scores_gemma":[0.008399905,0.0008260171,0.001693354,0.003766373,0.0007627168,0.004662239,0.001830896,0.001474076,0.004041398],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00080856,"about_ca_system_score_gemma":0.001783788,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004493375,"about_ca_topic_score_gemma":0.004311624,"domain_scores_codex":[0.997603,0.0008408806,0.0001906023,0.0007038505,0.0005466032,0.0001150582],"domain_scores_gemma":[0.9945262,0.00371179,0.0001204885,0.000956977,0.0006001152,0.00008437382],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001425449,0.0002232046,0.0006813965,0.00157422,0.00008734602,0.00005072029,0.0001523313,0.009759792,0.003503313,0.009733946,0.01261918,0.9614719],"study_design_scores_gemma":[0.0001403815,0.0006728052,0.003699697,0.000729976,0.0004717938,0.001300424,0.000537738,0.6612309,0.02712924,0.1089944,0.1949103,0.0001824596],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.01206602,0.1213773,0.84824,0.001352795,0.0004776041,0.0002710653,0.001127873,0.008126586,0.006960619],"genre_scores_gemma":[0.1971209,0.1088337,0.667954,0.001283426,0.001935699,0.0005811311,0.007762857,0.002125868,0.0124024],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.006974498,"threshold_uncertainty_score":0.023332,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1607836018395911,"score_gpt":0.4509311062684964,"score_spread":0.2901475044289052,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}