{"id":"W4411005413","doi":"10.26434/chemrxiv-2025-8h5q7","title":"MOF-ChemUnity: Unifying metal-organic framework data using large language models","year":2025,"lang":"en","type":"preprint","venue":"ChemRxiv","topic":"Metal-Organic Frameworks: Synthesis and Applications","field":"Chemistry","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada; University of Toronto","funders":"National Research Council Canada; Natural Sciences and Engineering Research Council of Canada; University of Toronto; Canada First Research Excellence Fund; Concordia University","keywords":"Metal-organic framework; Computer science; Chemistry; Organic chemistry","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023479,0.001572327,0.0008004092,0.005361152,0.0009871405,0.002153046,0.002362208,0.001892798,0.004288381],"category_scores_gemma":[0.01680955,0.0006400513,0.002906873,0.002791352,0.000831842,0.004149033,0.003417211,0.002052185,0.002215613],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001657449,"about_ca_system_score_gemma":0.003148109,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02582367,"about_ca_topic_score_gemma":0.0534302,"domain_scores_codex":[0.9982616,0.0005411467,0.0001587274,0.000561415,0.0003881018,0.00008902393],"domain_scores_gemma":[0.9934934,0.004500184,0.0003760544,0.0009946997,0.0004825537,0.0001531906],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000817466,0.0005612302,0.01237627,0.002888013,0.000847773,0.001592624,0.001342515,0.2370363,0.01339664,0.07345398,0.1141248,0.5415624],"study_design_scores_gemma":[0.0001193478,0.0001097166,0.001214907,0.0002060364,0.0001363632,0.0002245382,0.000248102,0.8022711,0.008284948,0.1062986,0.08077171,0.0001145427],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03817409,0.003330611,0.820352,0.003190786,0.0003643639,0.000585633,0.06436623,0.06302036,0.006615943],"genre_scores_gemma":[0.1949963,0.001358315,0.6989431,0.001049706,0.0001405536,0.0007308486,0.09823511,0.001977947,0.002568033],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02582367,"threshold_uncertainty_score":0.05134672,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07202616263590925,"score_gpt":0.3200032488494822,"score_spread":0.2479770862135729,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}