{"id":"W2341199921","doi":"10.1136/oemed-2015-103152","title":"Computer-based coding of free-text job descriptions to efficiently identify occupations in epidemiological studies","year":2016,"lang":"en","type":"article","venue":"Occupational and Environmental Medicine","topic":"Occupational and environmental lung diseases","field":"Medicine","cited_by":62,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"National Institutes of Health","keywords":"Computer science; Coding (social sciences); Terminology; Artificial intelligence; Machine learning; Data mining; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0381211,0.001455017,0.0009466056,0.01361099,0.000860414,0.001782123,0.001947224,0.0007674851,0.01015132],"category_scores_gemma":[0.137327,0.0007441753,0.001126484,0.007659794,0.00108631,0.00182839,0.00295991,0.001090021,0.003862534],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001732436,"about_ca_system_score_gemma":0.003906897,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005393076,"about_ca_topic_score_gemma":0.01205627,"domain_scores_codex":[0.958282,0.02657987,0.006780296,0.002692533,0.005033313,0.000632024],"domain_scores_gemma":[0.8242281,0.1001421,0.02849717,0.01463088,0.0306395,0.001862275],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.003317449,0.0009711426,0.2234796,0.006415801,0.0004525742,0.00037585,0.01029316,0.007688945,0.01199036,0.004398605,0.0439144,0.6867021],"study_design_scores_gemma":[0.001321282,0.00234894,0.7285076,0.005461389,0.0003835128,0.001272553,0.01068549,0.12494,0.02404342,0.01369656,0.08674326,0.0005959714],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.4234402,0.001927399,0.4910071,0.001254485,0.001079271,0.0311192,0.02579871,0.005799287,0.01857444],"genre_scores_gemma":[0.3160085,0.0007795443,0.6434696,0.0002579581,0.0002079753,0.02156578,0.01295709,0.0004623109,0.004291257],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0381211,"threshold_uncertainty_score":0.2016061,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06200902610992798,"score_gpt":0.3360384187649715,"score_spread":0.2740293926550436,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}