{"id":"W4401597819","doi":"10.1371/journal.pone.0307741","title":"GPT-4 as an X data annotator: Unraveling its performance on a stance classification task","year":2024,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Topic Modeling","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University; Lakehead University","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Artificial intelligence; Natural language processing; Task (project management); Annotation; Machine learning; Context (archaeology); Benchmark (surveying); Generalizability theory; Set (abstract data type); Psychology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01039837,0.002569637,0.00141679,0.001574035,0.00165292,0.002601253,0.002968923,0.003230641,0.005182195],"category_scores_gemma":[0.0280504,0.0006797156,0.001297283,0.001163149,0.001356639,0.00500297,0.005107576,0.004250753,0.007094215],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001489081,"about_ca_system_score_gemma":0.002379581,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008487102,"about_ca_topic_score_gemma":0.01351957,"domain_scores_codex":[0.9932401,0.0030269,0.0003290556,0.002146758,0.0008466817,0.000410464],"domain_scores_gemma":[0.9840307,0.008427859,0.0005820324,0.003378482,0.002868545,0.0007122591],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00407695,0.001040163,0.02100789,0.002098484,0.000521515,0.001216928,0.004736851,0.04014061,0.06303895,0.004840044,0.07873989,0.7785417],"study_design_scores_gemma":[0.0005014479,0.001438933,0.009301189,0.0003606904,0.0002461095,0.0009464209,0.002758706,0.8728382,0.05990639,0.0111998,0.04023098,0.0002711302],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4387319,0.003817672,0.4307419,0.002791385,0.002742532,0.001550702,0.0054566,0.08962689,0.02454044],"genre_scores_gemma":[0.5495991,0.0005775729,0.4153382,0.001823156,0.0002709767,0.001269359,0.01376785,0.003079489,0.01427438],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01039837,"threshold_uncertainty_score":0.0549925,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1955414476420611,"score_gpt":0.3002820790635207,"score_spread":0.1047406314214596,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}