{"id":"W4414973489","doi":"10.1109/icmla66185.2025.00227","title":"NLD-LLM: A systematic framework for evaluating small language transformer models on natural language description","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Natural language; Transformer; Language model; Natural language understanding; Source code; Process (computing); Iterative and incremental development; Set (abstract data type); Task (project management)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02491151,0.003180841,0.00110827,0.005510787,0.0009663464,0.003118492,0.003575036,0.001995812,0.003415643],"category_scores_gemma":[0.1157045,0.001058339,0.00211231,0.002407348,0.002430457,0.006092513,0.006273867,0.003918905,0.0009644061],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00382476,"about_ca_system_score_gemma":0.00556292,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008011717,"about_ca_topic_score_gemma":0.01403874,"domain_scores_codex":[0.9786023,0.01286891,0.001970804,0.001727012,0.004369223,0.0004617875],"domain_scores_gemma":[0.9103602,0.06791096,0.003970986,0.01055586,0.006202122,0.0009998964],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00204666,0.002506025,0.01822734,0.006001499,0.001101233,0.0003359201,0.001743828,0.5068885,0.02050854,0.05149113,0.02615051,0.3629988],"study_design_scores_gemma":[0.0003324775,0.001377572,0.001838667,0.000342086,0.0001626681,0.0001320282,0.0004550602,0.9415239,0.01860461,0.02624589,0.008864224,0.0001209482],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0848055,0.001461839,0.8773144,0.0007913876,0.0002527845,0.003951063,0.006041118,0.01917794,0.006203986],"genre_scores_gemma":[0.2585928,0.0004326203,0.7265422,0.000316173,0.00002661871,0.00363639,0.008257081,0.001340477,0.0008556864],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02491151,"threshold_uncertainty_score":0.1317462,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03913978433267463,"score_gpt":0.3440755121825993,"score_spread":0.3049357278499247,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}