{"id":"W3092530885","doi":"10.1088/2632-2153/abedc8","title":"Olympus: a benchmarking framework for noisy optimization and experiment planning","year":2021,"lang":"en","type":"article","venue":"Machine Learning Science and Technology","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":62,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Institute for Advanced Research; University of British Columbia; Vector Institute; University of Toronto","funders":"Office of Naval Research; Natural Sciences and Engineering Research Council of Canada; Defense Advanced Research Projects Agency","keywords":"Benchmarking; Suite; Benchmark (surveying); Python (programming language); Probabilistic logic; Software; Task (project management); Automation","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0110487,0.002727282,0.00188813,0.002143229,0.0008625079,0.002789351,0.005616001,0.002301869,0.009486179],"category_scores_gemma":[0.02762949,0.001428203,0.002272832,0.002052594,0.001961042,0.002772414,0.00376023,0.003649176,0.003723525],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002068934,"about_ca_system_score_gemma":0.00507632,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007409745,"about_ca_topic_score_gemma":0.007864609,"domain_scores_codex":[0.99401,0.002440051,0.0005602068,0.0007984454,0.001787918,0.0004032875],"domain_scores_gemma":[0.9903817,0.005099048,0.0005332404,0.002160918,0.001470846,0.0003542262],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005536387,0.000320204,0.003461021,0.001258105,0.0004358256,0.0002051764,0.0001634299,0.7828991,0.005199231,0.05572995,0.05898877,0.09078544],"study_design_scores_gemma":[0.0001204206,0.00009309495,0.0005395723,0.0000988264,0.00002642328,0.00004878364,0.00002297091,0.9482886,0.004461674,0.03000115,0.01624289,0.00005562263],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.008346026,0.0009780509,0.8889402,0.0004900196,0.0002358888,0.0004190079,0.004645281,0.08852451,0.007421003],"genre_scores_gemma":[0.1533981,0.0008758062,0.7988263,0.0008617151,0.0001128887,0.002504092,0.01909053,0.0214202,0.002910353],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0110487,"threshold_uncertainty_score":0.05843174,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01001838752986364,"score_gpt":0.2885465198786015,"score_spread":0.2785281323487379,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}