{"id":"W4409039368","doi":"10.2139/ssrn.5200113","title":"Optimizing Large Vision-Language Models for Context-Aware Construction Safety Assessment","year":2025,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Occupational Health and Safety Research","field":"Health Professions","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Context (archaeology); Computer science; Natural language processing; History; Archaeology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001000109,0.001682445,0.001539593,0.001175726,0.0007924462,0.002142704,0.002596081,0.002504337,0.00585609],"category_scores_gemma":[0.004606173,0.001396185,0.002243753,0.001209415,0.0006808363,0.002926874,0.001974631,0.002545873,0.002391289],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002121317,"about_ca_system_score_gemma":0.002484756,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03204772,"about_ca_topic_score_gemma":0.05010038,"domain_scores_codex":[0.9991436,0.0001970591,0.00004290034,0.0003020051,0.0001645232,0.0001499635],"domain_scores_gemma":[0.998046,0.001211814,0.00008834863,0.0002228068,0.0003088503,0.0001220953],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003863815,0.0003612352,0.001883734,0.0001516741,0.0001377226,0.0001608608,0.00009211466,0.8328989,0.009932618,0.002425502,0.005368455,0.1462008],"study_design_scores_gemma":[0.00001023174,0.00001585114,0.00008371541,0.000003418187,0.00001319562,0.000008900872,0.00001579883,0.9968359,0.00112155,0.001659227,0.0002276902,0.000004595767],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1230106,0.0008219267,0.8468227,0.0006507543,0.0002112786,0.0002142501,0.001498587,0.02272015,0.004049715],"genre_scores_gemma":[0.7456072,0.0002666068,0.2455832,0.0004455095,0.00009103503,0.0002069354,0.002630419,0.001555205,0.003613934],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03204772,"threshold_uncertainty_score":0.06372237,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04157353761623455,"score_gpt":0.4781213260617214,"score_spread":0.4365477884454868,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}