{"id":"W4396767426","doi":"10.1016/j.autcon.2024.105458","title":"Data-driven automatic classification model for construction accident cases using natural language processing with hyperparameter tuning","year":2024,"lang":"en","type":"article","venue":"Automation in Construction","topic":"Occupational Health and Safety Research","field":"Health Professions","cited_by":29,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"Korea Institute for Advancement of Technology; Ministry of Trade, Industry and Energy","keywords":"Hyperparameter; Accident (philosophy); Computer science; Artificial intelligence; Machine learning; Natural language processing; Data mining","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000923153,0.001160498,0.0007276814,0.001432226,0.0004436632,0.001230728,0.00153967,0.001112716,0.00269726],"category_scores_gemma":[0.003217841,0.000489309,0.001308061,0.0007033123,0.0002898307,0.001043675,0.0007616999,0.001451857,0.001778492],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001257807,"about_ca_system_score_gemma":0.001178285,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01136154,"about_ca_topic_score_gemma":0.01046386,"domain_scores_codex":[0.9993185,0.00009796225,0.00007049223,0.0003169564,0.0001099359,0.00008613314],"domain_scores_gemma":[0.998659,0.0006431794,0.00008620035,0.0001148323,0.0004563779,0.00004028901],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008288021,0.0008910419,0.01030227,0.0002369474,0.0002056682,0.0007612333,0.0002102022,0.4667465,0.02536203,0.001928314,0.01320156,0.4793255],"study_design_scores_gemma":[0.000008408576,0.00001431734,0.0004582398,0.000005850582,0.00001529914,0.00002483972,0.00001397489,0.996222,0.002156728,0.0007409006,0.0003332087,0.000006285708],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1699187,0.0004134377,0.8068799,0.0006377897,0.00017945,0.0004061599,0.003337406,0.01615645,0.002070651],"genre_scores_gemma":[0.7909967,0.0001305306,0.1966661,0.000186849,0.00005989416,0.0007009395,0.007939273,0.0002796378,0.003040145],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01136154,"threshold_uncertainty_score":0.02259082,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2022277656843931,"score_gpt":0.4993653301654243,"score_spread":0.2971375644810312,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}