{"id":"W4410818076","doi":"10.36227/techrxiv.174495034.42657551/v2","title":"LLM-in-the-Loop: Replicating Human Insight with LLMs for Better Machine Learning Applications","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Loop (graph theory); Computer science; Human-in-the-loop; Psychology; Artificial intelligence; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0201418,0.001338971,0.00147596,0.00287701,0.002257406,0.008902257,0.005679866,0.004828516,0.008861592],"category_scores_gemma":[0.07798557,0.0008484057,0.001552889,0.002952395,0.009077863,0.01864617,0.01523195,0.005614483,0.004677589],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003193826,"about_ca_system_score_gemma":0.004925032,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002254813,"about_ca_topic_score_gemma":0.002866064,"domain_scores_codex":[0.9811025,0.01182938,0.0007438568,0.002785825,0.002994143,0.0005443514],"domain_scores_gemma":[0.948137,0.02615787,0.002685112,0.01837884,0.003465828,0.001175341],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003138094,0.0003148085,0.003529609,0.00190151,0.0002269163,0.0002255646,0.004081588,0.03016535,0.006846438,0.5335232,0.02695963,0.3919116],"study_design_scores_gemma":[0.0000670923,0.0001234996,0.0005584835,0.0003089429,0.00004994751,0.000134972,0.0006686952,0.1177099,0.004241919,0.8110439,0.06501558,0.00007711693],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.007690917,0.002657765,0.9581407,0.01163609,0.0004568034,0.0002849221,0.0003166165,0.004575665,0.01424052],"genre_scores_gemma":[0.2956374,0.001962453,0.6865376,0.005239757,0.0009349976,0.0009336183,0.0007686057,0.001619789,0.006365859],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0201418,"threshold_uncertainty_score":0.1065212,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01782143569512096,"score_gpt":0.3059178948561969,"score_spread":0.288096459161076,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}