{"id":"W4400218600","doi":"10.2196/59641","title":"Large Language Models Can Enable Inductive Thematic Analysis of a Social Media Corpus in a Single Prompt: Human Validation Study","year":2024,"lang":"en","type":"article","venue":"JMIR Infodemiology","topic":"Misinformation and Its Impacts","field":"Social Sciences","cited_by":35,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institute on Drug Abuse; National Institutes of Health; National Cancer Institute; Research to Prevent Blindness","keywords":"Social media; Computer science; Thematic analysis; Thematic map; Natural language processing; Psychology; Linguistics; Artificial intelligence; Sociology; World Wide Web; Geography; Qualitative research; Cartography","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1006497,0.001925954,0.001006739,0.002384524,0.003114563,0.004308489,0.003773618,0.002848801,0.012284],"category_scores_gemma":[0.2913128,0.001138138,0.001557696,0.002086953,0.004151301,0.005756605,0.00775874,0.003520545,0.005499217],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002420821,"about_ca_system_score_gemma":0.003998677,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002951738,"about_ca_topic_score_gemma":0.005348918,"domain_scores_codex":[0.9114868,0.07138173,0.003173265,0.007931483,0.00493986,0.001086877],"domain_scores_gemma":[0.4422832,0.480625,0.007332667,0.03904802,0.02917307,0.001538118],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.004632141,0.005901874,0.05482744,0.006892966,0.0006584497,0.002615851,0.1887781,0.02253114,0.05395168,0.03841137,0.0497775,0.5710216],"study_design_scores_gemma":[0.00471168,0.004753047,0.05797892,0.003820264,0.0008659814,0.002639107,0.07534327,0.3024557,0.1201352,0.1371623,0.2889004,0.001234295],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3012355,0.0006895578,0.6530643,0.002061445,0.00067292,0.01851226,0.005320824,0.002991747,0.01545145],"genre_scores_gemma":[0.4009151,0.0001981697,0.5500189,0.001770422,0.0002427822,0.03575777,0.006609819,0.001042118,0.003444976],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8993503,"threshold_uncertainty_score":0.5322931,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08916595273790014,"score_gpt":0.405704456337404,"score_spread":0.3165385035995039,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}