{"id":"W3035066987","doi":"10.2196/16422","title":"Occupation Coding of Job Titles: Iterative Development of an Automated Coding Algorithm for the Canadian National Occupation Classification (ACA-NOC)","year":2020,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Health, Environment, Cognitive Aging","field":"Environmental Science","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; St. Michael's Hospital; University of Alberta; University of New Brunswick","funders":"WorkSafe Victoria","keywords":"Coding (social sciences); Computer science; Algorithm; Benchmark (surveying); Search algorithm; Data mining; Information retrieval; Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02147651,0.001653846,0.0009434848,0.01013875,0.00308637,0.002855936,0.002977483,0.0008386448,0.007741386],"category_scores_gemma":[0.1079283,0.0007335197,0.001336567,0.007957149,0.001511783,0.001624535,0.004158755,0.001803786,0.002745772],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.009741778,"about_ca_system_score_gemma":0.03859903,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.355892,"about_ca_topic_score_gemma":0.4416353,"domain_scores_codex":[0.9863311,0.00389289,0.002255499,0.002494704,0.004389315,0.0006364385],"domain_scores_gemma":[0.9341155,0.01801867,0.003172163,0.003222068,0.04068695,0.0007846036],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003953331,0.0002763097,0.03797223,0.002260625,0.0001601924,0.0002799894,0.01561421,0.01124619,0.01019229,0.01613993,0.07034664,0.835116],"study_design_scores_gemma":[0.0003746348,0.0003577576,0.09381063,0.002461376,0.0003690226,0.0006056701,0.02301328,0.5412951,0.03688292,0.03776998,0.2623512,0.0007083972],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0492537,0.0003903617,0.9113083,0.001035995,0.0003000337,0.01060318,0.01077187,0.007435034,0.00890149],"genre_scores_gemma":[0.03456687,0.0001017831,0.9532118,0.0001017782,0.0000188541,0.004733375,0.005081464,0.0004133311,0.001770699],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.644108,"threshold_uncertainty_score":0.707641,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1304285435754554,"score_gpt":0.4237081831491167,"score_spread":0.2932796395736613,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}