{"id":"W4412544568","doi":"10.2139/ssrn.5163979","title":"Retrieval-Augmented Generation for Natural Language Processing: A Survey","year":2025,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Topic Modeling","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University","funders":"","keywords":"Natural (archaeology); Computer science; Natural language processing; Information retrieval; Geography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00274603,0.001191807,0.002341758,0.002797392,0.0005818199,0.002397069,0.003714774,0.001474842,0.007766072],"category_scores_gemma":[0.008488678,0.0008491271,0.001647781,0.003935373,0.000798591,0.004485541,0.002045312,0.001388043,0.00449448],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007666456,"about_ca_system_score_gemma":0.001774081,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00379632,"about_ca_topic_score_gemma":0.00362838,"domain_scores_codex":[0.9975973,0.0008196788,0.0002041672,0.0006500052,0.0006134098,0.0001154563],"domain_scores_gemma":[0.9944721,0.003554019,0.0001307269,0.001125674,0.0006359927,0.00008142818],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001373666,0.0001977512,0.0006113939,0.001649396,0.00007329248,0.00005650354,0.0001538541,0.009584587,0.003193516,0.01096844,0.01278234,0.9605915],"study_design_scores_gemma":[0.0001475089,0.0007615052,0.003749588,0.0008358443,0.0004533812,0.001629678,0.0005623301,0.5895298,0.03049648,0.127242,0.2443995,0.0001924176],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.01234505,0.1148778,0.8507466,0.001179606,0.0004975816,0.0003185725,0.001245575,0.009900916,0.008888321],"genre_scores_gemma":[0.1840461,0.1041346,0.6846216,0.001196139,0.001594306,0.0006559473,0.007921881,0.002414299,0.01341519],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.007766072,"threshold_uncertainty_score":0.02598011,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02733157135005905,"score_gpt":0.3038457472400465,"score_spread":0.2765141758899874,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}