{"id":"W4366339700","doi":"10.1093/annweh/wxad020","title":"Evaluation of the updated SOCcer v2 algorithm for coding free-text job descriptions in three epidemiologic studies","year":2023,"lang":"en","type":"article","venue":"Annals of Work Exposures and Health","topic":"Occupational and environmental lung diseases","field":"Medicine","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"Center for Information Technology; National Cancer Institute; National Institutes of Health","keywords":"Intraclass correlation; Coding (social sciences); Kappa; Computer science; Numerical digit; Cohen's kappa; Algorithm; Artificial intelligence; Statistics; Machine learning; Medicine; Mathematics; Arithmetic; Psychometrics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00196253,0.00006391571,0.0002448053,0.00005622766,0.00009577528,0.000001815343,0.00004511438,0.0000311237,0.00001707714],"category_scores_gemma":[0.0002766636,0.00004089746,0.00006689438,0.0001996967,0.00008970877,0.00002988879,0.00004503683,0.00004627967,9.370196e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002714745,"about_ca_system_score_gemma":0.00006971099,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001196767,"about_ca_topic_score_gemma":0.0001589581,"domain_scores_codex":[0.999082,0.0001017424,0.0003180378,0.0001148793,0.0002389663,0.0001444178],"domain_scores_gemma":[0.9994397,0.0001467197,0.0001338684,0.0001235333,0.0001120058,0.00004415926],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00007362692,0.00009821313,0.8749083,0.000142904,0.0001038909,2.111701e-7,0.0002848875,0.0001942343,0.00003361755,0.0005766848,0.03093905,0.09264435],"study_design_scores_gemma":[0.0005714791,0.0002301538,0.9808044,0.0002482674,0.00005321696,3.679528e-7,0.0004323809,0.001797899,0.00004209263,0.01563299,0.0001526302,0.00003407011],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9665495,0.01213309,0.00007208751,0.02033484,0.00008693211,0.0006664582,0.0000925557,0.00001168718,0.00005282143],"genre_scores_gemma":[0.9945219,0.003629643,0.0005669363,0.001073322,0.00004789,0.0000549537,0.00004668331,0.000005425952,0.00005320273],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1058961,"threshold_uncertainty_score":0.166775,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4145771322248112,"score_gpt":0.4706850832498696,"score_spread":0.05610795102505833,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}