{"id":"W4321648896","doi":"10.48550/arxiv.2302.10986","title":"The FluidFlower International Benchmark Study: Process, Modeling Results, and Comparison to Experimental Data","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reservoir Engineering and Simulation Methods","field":"Engineering","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Nederlandse Organisatie voor Wetenschappelijk Onderzoek; Deutsche Forschungsgemeinschaft; Energi Simulation; National Science Foundation","keywords":"Benchmarking; Benchmark (surveying); Computer science; Process (computing); Ranking (information retrieval); Software deployment; Scale (ratio); Experimental data; Petrophysics; Data mining; Machine learning; Engineering; Statistics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005632578,0.001224703,0.0008072277,0.001402851,0.0009412479,0.001446391,0.001360078,0.001183626,0.0008346559],"category_scores_gemma":[0.006578195,0.000257852,0.0006973273,0.001594419,0.0007426749,0.00156196,0.0008765595,0.0009562902,0.0002344683],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00151476,"about_ca_system_score_gemma":0.001248322,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0156067,"about_ca_topic_score_gemma":0.0114355,"domain_scores_codex":[0.9982144,0.000529016,0.0001672431,0.000323526,0.000592014,0.0001738884],"domain_scores_gemma":[0.9961851,0.001450735,0.0003405268,0.0007714401,0.001084049,0.0001682096],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009126901,0.003316878,0.05076094,0.0003675424,0.0001475851,0.0004584369,0.0006122511,0.8467911,0.02275753,0.00394422,0.006722359,0.06320857],"study_design_scores_gemma":[0.0002514716,0.001610165,0.03157262,0.00005531668,0.00004545108,0.0001184107,0.0004765868,0.876437,0.08244541,0.001418099,0.005418282,0.0001511755],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.972317,0.000254092,0.01555648,0.0002612528,0.00007000323,0.0004613026,0.005395315,0.0008899485,0.0047946],"genre_scores_gemma":[0.9671017,0.0001740275,0.02373847,0.00003601436,0.00001698092,0.0003243578,0.007779811,0.0001143572,0.0007144148],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0156067,"threshold_uncertainty_score":0.03103173,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2038416776683454,"score_gpt":0.2954768178291744,"score_spread":0.09163514016082897,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}