{"id":"W4404875179","doi":"10.1016/j.engappai.2024.109490","title":"Low-cost language models: Survey and performance evaluation on Python code generation","year":2024,"lang":"en","type":"article","venue":"Engineering Applications of Artificial Intelligence","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"Novelis (Canada)","funders":"","keywords":"Computer science; Python (programming language); Programming language; Code generation; Operating system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007691982,0.00009503891,0.00008316852,0.0001550673,0.00005317941,0.000107384,0.0002598403,0.00004340554,0.000004024449],"category_scores_gemma":[0.00004970008,0.0001000264,0.00001955202,0.0004027976,0.00001731463,0.0002851279,0.00004800108,0.0001085871,0.00002384395],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004700795,"about_ca_system_score_gemma":0.0000422333,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003364143,"about_ca_topic_score_gemma":0.00002041273,"domain_scores_codex":[0.9990436,0.0000248123,0.0002594803,0.00032044,0.0002322794,0.0001193582],"domain_scores_gemma":[0.999315,0.0001154165,0.0000324299,0.0003961879,0.0001024322,0.00003859088],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[9.484536e-7,0.0000124212,0.000004412009,0.00002121117,0.000003144995,1.256617e-7,0.0003837296,0.5126597,0.003479976,0.05036129,0.000005362788,0.4330676],"study_design_scores_gemma":[0.000006128013,0.00001602112,0.0001067335,0.00003599745,0.000003307823,0.000001209557,0.00001435654,0.9454017,0.0535135,0.0007706509,0.00004122077,0.00008919361],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1460058,0.0002834786,0.8530039,0.00006461303,0.0001391031,0.0003062587,0.000007862299,0.0001333125,0.00005560799],"genre_scores_gemma":[0.9763111,0.00005103811,0.02331907,0.00001009613,0.00009394842,0.0001730725,0.00001755407,0.00001059197,0.00001346697],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8303053,"threshold_uncertainty_score":0.4078959,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08149017670230732,"score_gpt":0.3154793380580967,"score_spread":0.2339891613557894,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}