{"id":"W4389161391","doi":"10.2139/ssrn.4617116","title":"Using Large Language Models to Support Thematic Analysis in Empirical Legal Studies","year":2023,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Artificial Intelligence in Law","field":"Social Sciences","cited_by":24,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Thematic analysis; Thematic map; Coding (social sciences); Computer science; Qualitative analysis; Empirical research; Quality (philosophy); Qualitative research; Data science; Natural language processing; Sociology; Epistemology; Social science; Geography; Cartography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04081104,0.001283347,0.001177825,0.007504051,0.00287532,0.00687898,0.003560032,0.001751216,0.006659496],"category_scores_gemma":[0.185343,0.001247769,0.002567152,0.005032581,0.002250625,0.0143284,0.005862734,0.003728355,0.001809685],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003538533,"about_ca_system_score_gemma":0.005032326,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007060736,"about_ca_topic_score_gemma":0.01462303,"domain_scores_codex":[0.9616955,0.03319474,0.001540245,0.001920288,0.001403884,0.0002453265],"domain_scores_gemma":[0.6104632,0.3670511,0.005564678,0.01072278,0.005097711,0.001100537],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001866653,0.002490511,0.05349442,0.003009766,0.002136296,0.001165491,0.03319515,0.1193692,0.006750723,0.2675401,0.02967604,0.4793056],"study_design_scores_gemma":[0.0001959777,0.00009124971,0.002734149,0.0002987351,0.000265251,0.0001395731,0.003550444,0.6988863,0.001574788,0.2843302,0.007851355,0.00008199894],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.06556291,0.0003118623,0.9209363,0.002428261,0.0001288522,0.0009050847,0.002834641,0.002695998,0.004196084],"genre_scores_gemma":[0.3973851,0.0001644521,0.5935341,0.000355071,0.00007797869,0.002232146,0.005115126,0.0005305955,0.000605464],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9591889,"threshold_uncertainty_score":0.2158319,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1935123808274758,"score_gpt":0.4858269344481921,"score_spread":0.2923145536207162,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}