{"id":"W4400543975","doi":"10.26434/chemrxiv-2024-7fwxv","title":"Automated electrosynthesis reaction mining with multimodal large language models (MLLMs)","year":2024,"lang":"en","type":"preprint","venue":"ChemRxiv","topic":"Service-Oriented Architecture and Web Services","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Institute for Advanced Research; Vector Institute; University of Toronto","funders":"Office of Science; Canada First Research Excellence Fund; University of Toronto; U.S. Department of Energy; University of Minnesota; Nanyang Technological University; Ministry of Education - Singapore; Canadian Institute for Advanced Research","keywords":"Electrosynthesis; Computer science; Natural language processing; Chemistry; Electrochemistry","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0002851947,0.0005095169,0.0004393753,0.0003253536,0.0001326823,0.0004115843,0.001388582,0.0003572126,0.00001137615],"category_scores_gemma":[0.000007053361,0.0004138438,0.0001672342,0.0006141205,0.00002429432,0.0002542668,0.001602261,0.0008976878,0.00008280676],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001034361,"about_ca_system_score_gemma":0.0002190801,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00031229,"about_ca_topic_score_gemma":0.0002395056,"domain_scores_codex":[0.9972727,0.00006985034,0.0003258595,0.001226939,0.0004747722,0.0006299227],"domain_scores_gemma":[0.9981123,0.000108569,0.0002082262,0.00130767,0.0001133017,0.000149973],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006306448,0.002234443,0.000749352,0.01219196,0.004850303,0.003084232,0.2513726,0.04991563,0.5310325,0.05153898,0.005770833,0.0866285],"study_design_scores_gemma":[0.000297624,0.00004836216,0.0001446846,0.0006177501,0.0001026175,0.00004532378,0.0003746616,0.9219183,0.07197858,0.003322399,0.0005212225,0.0006285065],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9325203,0.001331815,0.05004342,0.001264945,0.0008250146,0.0004769141,0.00001498349,0.005503336,0.008019255],"genre_scores_gemma":[0.9733731,0.00002507327,0.02539668,0.0005226698,0.0002703362,0.0001038456,0.00007709972,0.00007138796,0.0001598315],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8720027,"threshold_uncertainty_score":0.9998313,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.008829228375520934,"score_gpt":0.2410132929112305,"score_spread":0.2321840645357096,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}