{"id":"W2042110087","doi":"10.1021/ci600426e","title":"Evaluating Virtual Screening Methods:  Good and Bad Metrics for the “Early Recognition” Problem","year":2007,"lang":"en","type":"article","venue":"Journal of Chemical Information and Modeling","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":858,"is_retracted":false,"has_abstract":true,"ca_institutions":"TransCanada (Canada)","funders":"","keywords":"Computer science; Virtual screening; Artificial intelligence; Machine learning; Pattern recognition (psychology); Bioinformatics; Drug discovery; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0459551,0.003036061,0.003733692,0.00740708,0.001196741,0.00512961,0.002545368,0.003982814,0.001057059],"category_scores_gemma":[0.1572501,0.0006693782,0.001609953,0.003259364,0.004911251,0.006754202,0.004395242,0.004299588,0.0007020878],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001986866,"about_ca_system_score_gemma":0.002030683,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001275111,"about_ca_topic_score_gemma":0.0009975823,"domain_scores_codex":[0.9403703,0.03149493,0.004354765,0.004784047,0.01801507,0.0009809339],"domain_scores_gemma":[0.7741375,0.1860831,0.0116665,0.01384729,0.01183386,0.002431677],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001449659,0.0006448276,0.03724989,0.003737637,0.001449342,0.0003807419,0.000730237,0.2413212,0.01824119,0.06886213,0.01271938,0.6132137],"study_design_scores_gemma":[0.0001582149,0.003698953,0.0117234,0.0006446316,0.0004053448,0.00148023,0.0003810918,0.8216453,0.03914955,0.1072332,0.0130886,0.0003915508],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07267337,0.02385874,0.892325,0.003080142,0.0005142502,0.000473149,0.0005658962,0.00220498,0.004304479],"genre_scores_gemma":[0.515578,0.004861917,0.4733319,0.001774885,0.0005966791,0.0005880766,0.00124512,0.0006240394,0.001399388],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9540449,"threshold_uncertainty_score":0.2430367,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1342111082886273,"score_gpt":0.4229986251823157,"score_spread":0.2887875168936884,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}