{"id":"W7045818181","doi":"","title":"Benchmarking Foundation Evaluation Practices 2020","year":2020,"lang":"en","type":"dataset","venue":"Issue Lab (Candid)","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Benchmarking; Foundation (evidence); Function (biology); Investment (military); Survey data collection; Data collection; Evaluation methods","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01088662,0.001029989,0.0007345609,0.00731642,0.001139356,0.003376344,0.002382696,0.00109586,0.02120923],"category_scores_gemma":[0.03355142,0.0006771168,0.0009323717,0.0120378,0.0004212823,0.001746542,0.002947171,0.001570451,0.02119993],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004188669,"about_ca_system_score_gemma":0.005854541,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.06083626,"about_ca_topic_score_gemma":0.08354759,"domain_scores_codex":[0.9909104,0.003084105,0.001313616,0.001221562,0.002646799,0.0008235459],"domain_scores_gemma":[0.9796251,0.003903463,0.002178746,0.005227641,0.00800227,0.001062784],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001197463,0.00006354537,0.007242359,0.0005703631,0.00003319285,0.00002713222,0.0001510694,0.0005692336,0.0001090291,0.003073726,0.9706625,0.01737799],"study_design_scores_gemma":[0.000143223,0.00003027598,0.02195275,0.0003689252,0.00001615335,0.00004794544,0.0003204266,0.0009548995,0.00047433,0.0008776558,0.9747776,0.00003590817],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.003358173,0.0002234762,0.0006739733,0.0004314486,0.00006888053,0.0001558951,0.9824055,0.001075273,0.01160747],"genre_scores_gemma":[0.003532063,0.00009265656,0.001491098,0.0001005996,0.000008083603,0.0004223225,0.991152,0.0001268584,0.003074228],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.9891134,"threshold_uncertainty_score":0.1209643,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05436349869071153,"score_gpt":0.3736739790814738,"score_spread":0.3193104803907623,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}