{"id":"W4244573874","doi":"10.2196/preprints.16422","title":"Occupation Coding of Job Titles: Iterative Development of an Automated Coding Algorithm for the Canadian National Occupation Classification (ACA-NOC) (Preprint)","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Healthcare Systems and Public Health","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; St. Michael's Hospital; University of Alberta; University of New Brunswick","funders":"","keywords":"Coding (social sciences); Computer science; Algorithm; Benchmark (surveying); Search algorithm; Upload; Information retrieval; Data mining; Preprint; Matching (statistics); Mathematics; World Wide Web; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003032509,0.0002249607,0.0005164657,0.0004831879,0.0002452258,0.00005011565,0.0001902765,0.000466115,0.00009532063],"category_scores_gemma":[0.00052041,0.0001740207,0.00009700159,0.00021854,0.00004644054,0.000120851,0.0000562181,0.0003494273,0.00001046684],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001762374,"about_ca_system_score_gemma":0.02019531,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.08754932,"about_ca_topic_score_gemma":0.1106578,"domain_scores_codex":[0.9971457,0.000184903,0.001263581,0.0004527987,0.0006664242,0.0002865985],"domain_scores_gemma":[0.9955997,0.0003704603,0.0008173552,0.0004351005,0.002529429,0.0002479478],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003464998,0.0005294861,0.05819329,0.01941188,0.001305128,0.000002578848,0.04323648,0.001845489,0.002154871,0.04079952,0.01293008,0.8192447],"study_design_scores_gemma":[0.0006489676,0.000109508,0.2227944,0.001148401,0.00003468595,0.000005623678,0.001163716,0.769082,0.0009570922,0.00007348762,0.003785914,0.0001962637],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.102435,0.00009854534,0.8721256,0.002727689,0.001670686,0.01052627,0.0009739915,0.0002817237,0.009160512],"genre_scores_gemma":[0.9438749,0.00002066617,0.05298372,0.0002376464,0.0001907175,0.0002924191,0.002016518,0.00002845279,0.0003549652],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8414398,"threshold_uncertainty_score":0.9853593,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1113322150339254,"score_gpt":0.4109827966512818,"score_spread":0.2996505816173564,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}