{"id":"W4400543975","doi":"10.26434/chemrxiv-2024-7fwxv","title":"Automated electrosynthesis reaction mining with multimodal large language models (MLLMs)","year":2024,"lang":"en","type":"preprint","venue":"ChemRxiv","topic":"Service-Oriented Architecture and Web Services","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Institute for Advanced Research; Vector Institute; University of Toronto","funders":"Office of Science; Canada First Research Excellence Fund; University of Toronto; U.S. Department of Energy; University of Minnesota; Nanyang Technological University; Ministry of Education - Singapore; Canadian Institute for Advanced Research","keywords":"Electrosynthesis; Computer science; Natural language processing; Chemistry; Electrochemistry","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003516775,0.002304662,0.0009494157,0.003160527,0.0007049194,0.003183879,0.002320527,0.00167422,0.006588966],"category_scores_gemma":[0.01128627,0.00093282,0.003787067,0.001860271,0.0007920733,0.003720014,0.002603261,0.002472593,0.005734639],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001503173,"about_ca_system_score_gemma":0.002461687,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004284189,"about_ca_topic_score_gemma":0.008191996,"domain_scores_codex":[0.9977291,0.0007316596,0.0002287674,0.0006739704,0.0005212144,0.0001153399],"domain_scores_gemma":[0.9942116,0.00390572,0.0004418168,0.0006938005,0.0006307466,0.000116191],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009968391,0.0004698015,0.008230507,0.004089534,0.0008335086,0.00131118,0.001014658,0.1152737,0.07385556,0.03052466,0.07196425,0.6914358],"study_design_scores_gemma":[0.00008178392,0.0001621524,0.001278194,0.0001786918,0.0001654272,0.0003953697,0.0003354594,0.8294755,0.06549168,0.05282306,0.04947997,0.0001326937],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.024453,0.001914975,0.8567328,0.00183351,0.0002300228,0.0004384833,0.02041337,0.08946604,0.004517957],"genre_scores_gemma":[0.1405862,0.001172187,0.8130164,0.0008973599,0.0001233348,0.0009272478,0.03586831,0.003059553,0.004349331],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006588966,"threshold_uncertainty_score":0.02204227,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.008829228375520934,"score_gpt":0.2410132929112305,"score_spread":0.2321840645357096,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}