{"id":"W7134171330","doi":"10.1109/bigdata66926.2025.11401680","title":"From Text to Insight: Towards Robust RAG Pipelines for Transcript-Based Clinical Screening","year":2025,"lang":"","type":"article","venue":"","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Pipeline transport; Pipeline (software); Noise (video); Key (lock)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005084418,0.002547306,0.001597548,0.008401962,0.001173793,0.004629114,0.002431476,0.002306893,0.01001269],"category_scores_gemma":[0.02132017,0.0009549242,0.003388806,0.004641747,0.00104642,0.004518612,0.005504398,0.003001917,0.009144573],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001371404,"about_ca_system_score_gemma":0.003700761,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004805256,"about_ca_topic_score_gemma":0.008380889,"domain_scores_codex":[0.9966553,0.0007983821,0.0003650449,0.00107097,0.0008195467,0.0002907742],"domain_scores_gemma":[0.9895438,0.006269241,0.0007811699,0.001634067,0.001361075,0.0004107263],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001950141,0.0006071623,0.01176941,0.003320345,0.001049447,0.001716873,0.001227259,0.02089598,0.04767389,0.02838916,0.1334812,0.7479191],"study_design_scores_gemma":[0.0003331435,0.0005063468,0.009795624,0.0008532266,0.001058087,0.001760243,0.001384654,0.4933384,0.07416645,0.2409941,0.1755067,0.0003030511],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0152913,0.0021419,0.7966753,0.002495527,0.0003542013,0.0007026877,0.05376164,0.1243842,0.004193196],"genre_scores_gemma":[0.1210733,0.001528827,0.7690328,0.001686353,0.0003040729,0.0006827669,0.09651811,0.004693141,0.004480567],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01001269,"threshold_uncertainty_score":0.03349578,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0747946657671624,"score_gpt":0.3732849851336149,"score_spread":0.2984903193664524,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}