{"id":"W4414015757","doi":"10.11159/mvml25.113","title":"A Comparative Evaluation of Vision Language Models for Waste Classification in Few-Shot Settings","year":2025,"lang":"en","type":"article","venue":"Proceedings of the World Congress on Electrical Engineering and Computer Systems and Science","topic":"Municipal Solid Waste Management","field":"Environmental Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Shot (pellet); Computer science; Artificial intelligence; Natural language processing; One shot; Engineering; Materials science; Mechanical engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009970024,0.00007690947,0.000158229,0.0001561196,0.00006268334,0.00005053388,0.0002370797,0.00001656621,3.426158e-7],"category_scores_gemma":[0.00002975496,0.00005622895,0.00001548674,0.0008464695,0.0001010172,0.0001412146,0.0001512439,0.00006524439,6.798565e-8],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008508232,"about_ca_system_score_gemma":0.00001004501,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004082078,"about_ca_topic_score_gemma":0.000005214836,"domain_scores_codex":[0.99909,0.000008391706,0.0001976021,0.0002279724,0.0003373016,0.0001387453],"domain_scores_gemma":[0.9996834,0.00005993495,0.0001011394,0.00007193934,0.00005638918,0.00002717114],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003710749,0.00007615444,0.001113436,0.0002403613,0.00001307941,4.873001e-8,0.0008310069,0.903268,0.01944243,0.06528804,0.0004768042,0.009213481],"study_design_scores_gemma":[0.0002431441,0.00005171023,0.004496175,0.0002590427,0.0000102126,3.483083e-7,0.00007711961,0.9922805,0.00237172,0.00009606378,0.00006025205,0.0000536826],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9962261,0.0001224567,0.001276967,0.0001473953,0.0001764149,0.000629844,7.943392e-7,0.000008469206,0.001411503],"genre_scores_gemma":[0.9995767,0.000002638449,0.0002379788,0.00001618587,0.00000951029,0.00003948643,1.131623e-7,0.000002208874,0.0001151178],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.08901247,"threshold_uncertainty_score":0.229295,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02186614719294598,"score_gpt":0.2802667748184238,"score_spread":0.2584006276254778,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}