{"id":"W3217379076","doi":"10.1136/fmch-2021-001287","title":"Developing and testing an automated qualitative assistant (AQUA) to support qualitative analysis","year":2021,"lang":"en","type":"article","venue":"Family Medicine and Community Health","topic":"Qualitative Research Methods and Ethics","field":"Social Sciences","cited_by":47,"is_retracted":false,"has_abstract":true,"ca_institutions":"Centre for Family Medicine","funders":"Penn State College of Medicine; Huck Institutes of the Life Sciences; Pennsylvania State University; University of Pennsylvania","keywords":"Computer science; Latent Dirichlet allocation; Coding (social sciences); Machine learning; Information retrieval; Cluster analysis; Artificial intelligence; Transparency (behavior); Natural language processing; Data mining; Topic model; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","sts"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.06000815,0.000178243,0.0007207246,0.0003331247,0.003298583,0.0000857882,0.0002270111,0.0001053483,0.00002531952],"category_scores_gemma":[0.02344725,0.0001579322,0.00003773234,0.002928013,0.001254766,0.000218293,0.0001840483,0.001010434,0.000002410357],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002659375,"about_ca_system_score_gemma":0.002405965,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.1301646,"about_ca_topic_score_gemma":0.07128347,"domain_scores_codex":[0.9420547,0.0557674,0.0005726906,0.0003020036,0.0007414035,0.0005617744],"domain_scores_gemma":[0.9770414,0.0204252,0.0002054703,0.0003508516,0.001097285,0.0008797975],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.00001703182,0.00006202947,0.0007013925,0.0001461905,0.0001309464,0.0000068697,0.9418212,9.99013e-7,0.0001033074,0.04337528,0.0005085433,0.01312617],"study_design_scores_gemma":[0.0003020635,0.001012868,0.01889999,0.0002284288,0.00004965652,0.000001132633,0.958499,0.000201216,0.000005377684,0.01653529,0.004084141,0.000180882],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.886856,0.001120514,0.03256704,0.07196508,0.00007856725,0.0003611443,0.00006650504,0.0002276917,0.006757473],"genre_scores_gemma":[0.8673465,0.00121972,0.1191029,0.01182975,0.00009042401,0.00002591668,0.0001290502,0.0000186788,0.000237117],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.08653581,"threshold_uncertainty_score":0.997999,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7362791457125837,"score_gpt":0.6989436886705526,"score_spread":0.03733545704203112,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}