{"id":"W4391220979","doi":"10.1287/mksc.2023.0454","title":"Frontiers: Determining the Validity of Large Language Models for Automated Perceptual Analysis","year":2024,"lang":"en","type":"article","venue":"Marketing Science","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":131,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Perception; Computer science; Data science; Human language; Natural language processing; Language model; Artificial intelligence; Cognitive psychology; Machine learning; Econometrics; Economics; Psychology; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01432575,0.001313321,0.0009017388,0.002163933,0.001310867,0.004983549,0.001766333,0.001598563,0.006755377],"category_scores_gemma":[0.1096614,0.0006853343,0.001284355,0.001011143,0.002497862,0.009420702,0.003825568,0.002783271,0.002140907],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001343235,"about_ca_system_score_gemma":0.001764656,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004965283,"about_ca_topic_score_gemma":0.00350942,"domain_scores_codex":[0.9898239,0.007222359,0.0003348776,0.001171556,0.001129301,0.0003180266],"domain_scores_gemma":[0.8459277,0.1406275,0.0021106,0.007041455,0.003202262,0.001090534],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.005386849,0.001587588,0.03422331,0.001412561,0.0006016809,0.001102908,0.007221573,0.1685957,0.03119779,0.1194721,0.0244286,0.6047692],"study_design_scores_gemma":[0.0001405763,0.0002503211,0.002698902,0.00006518636,0.00004771851,0.0001576862,0.000760927,0.9137572,0.003607471,0.07575272,0.002711252,0.00005011332],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2810509,0.0009882619,0.6962207,0.001780699,0.000319998,0.0007059227,0.00192498,0.005604601,0.01140396],"genre_scores_gemma":[0.8772382,0.0001747562,0.119044,0.0002335979,0.0000911457,0.0004129232,0.001390314,0.0004601898,0.0009548303],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01432575,"threshold_uncertainty_score":0.07576275,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04878391879767375,"score_gpt":0.4075470857221072,"score_spread":0.3587631669244334,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}