{"id":"W4402335544","doi":"10.1101/2024.09.06.24313186","title":"Loon Lens 1.0 Validation: Agentic AI for Title and Abstract Screening in Systematic Literature Reviews","year":2024,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Systematic review; Lens (geology); Through-the-lens metering; Psychology; Data science; Political science; Computer science; MEDLINE; Law; Physics; Optics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3382286,0.004477708,0.01102864,0.03693046,0.003513279,0.01567787,0.005584319,0.004934898,0.05312922],"category_scores_gemma":[0.6467534,0.004792546,0.01800273,0.02865907,0.004064401,0.01180365,0.008327414,0.003395346,0.009438296],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00701594,"about_ca_system_score_gemma":0.04095824,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003579027,"about_ca_topic_score_gemma":0.01215393,"domain_scores_codex":[0.660534,0.188319,0.100144,0.02067204,0.02707361,0.003257219],"domain_scores_gemma":[0.2425662,0.5684946,0.07842966,0.03921609,0.06871958,0.002573972],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.008077057,0.0003861246,0.0277412,0.5292333,0.0243696,0.001186842,0.0102267,0.002632291,0.005720951,0.01405818,0.1652552,0.2111125],"study_design_scores_gemma":[0.02451051,0.003750338,0.06455503,0.2437792,0.07395767,0.003556306,0.006123499,0.03622419,0.013615,0.04280851,0.4840033,0.003116448],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.06490085,0.05045136,0.2960661,0.02239983,0.007735017,0.2800934,0.1564302,0.07721622,0.04470715],"genre_scores_gemma":[0.08671686,0.005531162,0.5828732,0.00370307,0.0009158144,0.2949849,0.01383342,0.006230012,0.005211561],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.6617714,"threshold_uncertainty_score":0.8160819,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1130685049969929,"score_gpt":0.4176119119838094,"score_spread":0.3045434069868165,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}