{"id":"W7130556940","doi":"10.1109/fllm67465.2025.11391221","title":"Emissions and Performance Trade-off Between Small and Large Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Sustainable development; Language model; Reduction (mathematics); Natural language; Sustainability","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004690957,0.0009435154,0.0006602404,0.0004979869,0.0003232164,0.00150005,0.00130642,0.001135003,0.002042061],"category_scores_gemma":[0.02036583,0.0004171817,0.0006216819,0.0003973097,0.0008311063,0.003999081,0.001281977,0.001950484,0.0006706662],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001155846,"about_ca_system_score_gemma":0.001335318,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006200273,"about_ca_topic_score_gemma":0.009383889,"domain_scores_codex":[0.9983608,0.0007353973,0.0001166478,0.0004199218,0.0002116177,0.0001556004],"domain_scores_gemma":[0.9848867,0.01244962,0.0003905585,0.001448297,0.0005483227,0.0002764209],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001546875,0.0005511869,0.008199118,0.0005582683,0.0002793098,0.0002016753,0.0004292398,0.7957354,0.02251623,0.007561384,0.002977428,0.1594439],"study_design_scores_gemma":[0.00007636494,0.0004906663,0.001541081,0.00003321477,0.00008081582,0.00006399188,0.000150771,0.9717611,0.01376181,0.01028341,0.00171895,0.00003795134],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8280122,0.002554725,0.1486567,0.002423604,0.0002172838,0.0001615492,0.0008312139,0.005477405,0.01166539],"genre_scores_gemma":[0.9584871,0.0003015735,0.0385887,0.000250763,0.00002555659,0.00008815321,0.0006013416,0.0002960857,0.001360676],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006200273,"threshold_uncertainty_score":0.02480847,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01514885999433322,"score_gpt":0.2757685991056981,"score_spread":0.2606197391113648,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}