{"id":"W7130556940","doi":"10.1109/fllm67465.2025.11391221","title":"Emissions and Performance Trade-off Between Small and Large Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Sustainable development; Language model; Reduction (mathematics); Natural language; Sustainability","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002252786,0.0004242288,0.0005963213,0.0002111372,0.0009392056,0.0007145051,0.0006319744,0.0002386458,0.001122999],"category_scores_gemma":[0.0002507226,0.0003597259,0.00004062547,0.0004090282,0.000479819,0.0006343941,0.001019739,0.0003389724,0.00003879962],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004014614,"about_ca_system_score_gemma":0.0001739033,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001302251,"about_ca_topic_score_gemma":0.00002915524,"domain_scores_codex":[0.9967375,0.0003366952,0.0006537518,0.001050068,0.0003320893,0.0008898435],"domain_scores_gemma":[0.9985152,0.0003346991,0.0001824969,0.0006013115,0.00003779061,0.0003284317],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002546688,0.0003918121,0.2690516,0.003862089,0.00009880017,0.00004499262,0.02221124,0.001737849,0.5809377,0.02547636,0.001981137,0.09395175],"study_design_scores_gemma":[0.00300442,0.0006372986,0.2821089,0.001623757,0.0003577438,0.0000539497,0.002961725,0.5864289,0.1082671,0.003772067,0.008850132,0.001933954],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9743685,0.001770537,0.009186774,0.0020478,0.0005616756,0.0004267132,0.00009138732,0.0001529526,0.01139366],"genre_scores_gemma":[0.9829822,0.0007221512,0.008128759,0.000534457,0.0001201803,0.00001474134,0.000006544498,0.00002219514,0.007468751],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.584691,"threshold_uncertainty_score":0.9998855,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01514885999433322,"score_gpt":0.2757685991056981,"score_spread":0.2606197391113648,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}