{"id":"W4307136538","doi":"10.1145/3517428.3544818","title":"Challenging and Improving Current Evaluation Methods for Colour Identification Aids","year":2022,"lang":"en","type":"article","venue":"","topic":"Visual perception and processing mechanisms","field":"Neuroscience","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Guelph","funders":"","keywords":"Identification (biology); Computer science; Task (project management); Population; Artificial intelligence; Everyday life; Disease; Machine learning; Medicine; Environmental health; Engineering; Pathology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07548492,0.002230596,0.001529918,0.004048489,0.001227237,0.007557996,0.00428232,0.002477549,0.005543068],"category_scores_gemma":[0.2725335,0.0007170516,0.001228125,0.00161105,0.001455499,0.007381399,0.003568698,0.001825778,0.002833805],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002210906,"about_ca_system_score_gemma":0.002712155,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002951245,"about_ca_topic_score_gemma":0.004205477,"domain_scores_codex":[0.9183199,0.04910651,0.007015411,0.004242939,0.02042573,0.0008895059],"domain_scores_gemma":[0.6816142,0.1970695,0.01117635,0.01820798,0.08937351,0.002558499],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.002008736,0.001042431,0.0156894,0.006226983,0.0004573909,0.0001453461,0.003085152,0.004598016,0.0192507,0.008069009,0.01287544,0.9265513],"study_design_scores_gemma":[0.001747561,0.00964285,0.1313979,0.01987047,0.002118328,0.003784114,0.01651622,0.3616686,0.1336517,0.1078936,0.2094227,0.002285898],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1050615,0.01642089,0.8408954,0.003842517,0.00165254,0.004696264,0.001213685,0.00451681,0.02170049],"genre_scores_gemma":[0.2977436,0.004065108,0.6871565,0.001400314,0.0002904453,0.003274582,0.00118652,0.0009106104,0.003972275],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.07548492,"threshold_uncertainty_score":0.3992072,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2189193401196005,"score_gpt":0.4906573913871914,"score_spread":0.271738051267591,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}