{"id":"W4319057825","doi":"10.1093/annweh/wxad002","title":"Automated Coding of Job Descriptions From a General Population Study: Overview of Existing Tools, Their Application and Comparison","year":2023,"lang":"en","type":"article","venue":"Annals of Work Exposures and Health","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"Saint John Regional Hospital; University of Toronto; University of New Brunswick","funders":"National Cancer Institute; National Institutes of Health; European Commission; University of Warwick","keywords":"Coding (social sciences); Computer science; Population; Statistics; Medicine; Mathematics; Environmental health","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05967029,0.0008281415,0.0007482658,0.01572848,0.001107086,0.001450604,0.002360372,0.0005814717,0.00148848],"category_scores_gemma":[0.1188542,0.0006031027,0.001096552,0.008036927,0.001411579,0.00172705,0.002762523,0.0006921827,0.0004832047],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002134018,"about_ca_system_score_gemma":0.004528118,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01299428,"about_ca_topic_score_gemma":0.0157993,"domain_scores_codex":[0.9241318,0.05018649,0.008335961,0.003977699,0.01255479,0.0008132564],"domain_scores_gemma":[0.8215058,0.09447197,0.02141093,0.01583522,0.04550131,0.001274643],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007724802,0.0007123145,0.4821728,0.002934637,0.0002688102,0.0003644092,0.02001239,0.001624701,0.00282553,0.001097468,0.003339155,0.4838753],"study_design_scores_gemma":[0.000190235,0.001868339,0.9496466,0.002195212,0.0002336528,0.001437741,0.01435743,0.01225482,0.004298186,0.001430986,0.01186424,0.0002225222],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.865344,0.00531373,0.1078899,0.0005530113,0.0002101908,0.008118547,0.004931331,0.0009911819,0.006647949],"genre_scores_gemma":[0.7537395,0.00397768,0.2222822,0.0002465667,0.0001514653,0.01062781,0.007309632,0.0002085274,0.001456551],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.05967029,"threshold_uncertainty_score":0.3155704,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7392532214113784,"score_gpt":0.5341445500158267,"score_spread":0.2051086713955517,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}