{"id":"W4414602263","doi":"10.1057/s41599-025-05834-4","title":"LLMs as annotators: the effect of party cues on labelling decisions by large language models","year":2025,"lang":"en","type":"article","venue":"Humanities and Social Sciences Communications","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"Universität Wien","keywords":"Politics; Statement (logic); Test (biology); Motivated reasoning; Replicate","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1063156,0.002000271,0.001454395,0.00160417,0.002431427,0.004842775,0.002078812,0.003110915,0.004189137],"category_scores_gemma":[0.5048245,0.001311302,0.001229977,0.001982674,0.00401463,0.007633348,0.006946012,0.00419678,0.002599139],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002205564,"about_ca_system_score_gemma":0.001849814,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005390391,"about_ca_topic_score_gemma":0.008509548,"domain_scores_codex":[0.8212453,0.1469591,0.006561264,0.01438835,0.009535736,0.001310303],"domain_scores_gemma":[0.3073663,0.5910538,0.03664166,0.04957658,0.01279397,0.002567654],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01814,0.00112678,0.3637046,0.004532148,0.004283752,0.001174742,0.05698952,0.03602263,0.0747926,0.02962983,0.03190995,0.3776935],"study_design_scores_gemma":[0.003174563,0.003621881,0.2372538,0.002056285,0.001854333,0.002711078,0.01412487,0.432648,0.08759615,0.1567976,0.05625394,0.001907676],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7353799,0.00256295,0.2234581,0.004935473,0.001152221,0.001594868,0.002020504,0.003041473,0.02585449],"genre_scores_gemma":[0.9304181,0.0002587017,0.06025815,0.002818095,0.0002148123,0.001228981,0.001482667,0.000888212,0.002432263],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8936844,"threshold_uncertainty_score":0.5622572,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07625001237173913,"score_gpt":0.3429990491010771,"score_spread":0.266749036729338,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}