{"id":"W4366339700","doi":"10.1093/annweh/wxad020","title":"Evaluation of the updated SOCcer v2 algorithm for coding free-text job descriptions in three epidemiologic studies","year":2023,"lang":"en","type":"article","venue":"Annals of Work Exposures and Health","topic":"Occupational and environmental lung diseases","field":"Medicine","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"Center for Information Technology; National Cancer Institute; National Institutes of Health","keywords":"Intraclass correlation; Coding (social sciences); Kappa; Computer science; Numerical digit; Cohen's kappa; Algorithm; Artificial intelligence; Statistics; Machine learning; Medicine; Mathematics; Arithmetic; Psychometrics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06200107,0.001416105,0.001073875,0.005220383,0.001139079,0.002623522,0.002844685,0.001248124,0.003059266],"category_scores_gemma":[0.1502346,0.0009712818,0.001743896,0.003296672,0.000634908,0.001392866,0.003029426,0.001240163,0.000712934],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002320824,"about_ca_system_score_gemma":0.00483294,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01774128,"about_ca_topic_score_gemma":0.01905472,"domain_scores_codex":[0.9561698,0.026307,0.006579591,0.005180713,0.005201843,0.0005610032],"domain_scores_gemma":[0.8484955,0.09291627,0.01188361,0.009309998,0.03618975,0.001204818],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.008076732,0.000801617,0.5226847,0.00159712,0.00215713,0.0002699232,0.003254883,0.04019358,0.005220771,0.002118152,0.01133088,0.4022945],"study_design_scores_gemma":[0.002763042,0.002937495,0.4360386,0.001048006,0.001598282,0.001366113,0.002614049,0.5086966,0.01871088,0.003489809,0.02015027,0.0005867879],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6730422,0.001015552,0.3005787,0.0005341562,0.0003702383,0.007164325,0.008768979,0.003666235,0.004859623],"genre_scores_gemma":[0.4353502,0.0003216258,0.547805,0.0001772052,0.00006097844,0.005972117,0.008301226,0.000629975,0.001381625],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.06200107,"threshold_uncertainty_score":0.3278969,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4145771322248112,"score_gpt":0.4706850832498696,"score_spread":0.05610795102505833,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}