{"id":"W4414158748","doi":"10.1007/978-981-95-0988-1_1","title":"Evaluating the Behavior of Small Language Models in Answering Binary Questions","year":2025,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Security token; Binary number; Natural language; Language model; Binary classification; Natural (archaeology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01844423,0.001649275,0.00177295,0.001527907,0.001010258,0.003425373,0.002520936,0.003849817,0.003850022],"category_scores_gemma":[0.09344564,0.0009257402,0.001139207,0.00141597,0.001335072,0.006770864,0.002011802,0.0035913,0.001497748],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002180859,"about_ca_system_score_gemma":0.001380838,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009245876,"about_ca_topic_score_gemma":0.009936208,"domain_scores_codex":[0.9891667,0.00768354,0.0004609504,0.001273317,0.001088019,0.0003275718],"domain_scores_gemma":[0.7515745,0.238919,0.001689023,0.004101399,0.002104457,0.001611648],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01127882,0.002953261,0.03541398,0.001187342,0.0008853013,0.0002692749,0.001743846,0.4903834,0.014343,0.01504628,0.02079019,0.4057054],"study_design_scores_gemma":[0.00008877247,0.0003003985,0.0009848687,0.00002360304,0.00007058053,0.0000367994,0.0001412053,0.9879686,0.002080821,0.007852118,0.0004347693,0.00001740317],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8266234,0.003763775,0.1510394,0.003476699,0.0003330355,0.0004313265,0.001270206,0.004418313,0.008643758],"genre_scores_gemma":[0.9169098,0.000530636,0.07543927,0.00058423,0.0001872644,0.0002088767,0.002557085,0.0004683609,0.003114395],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01844423,"threshold_uncertainty_score":0.09754354,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1177777415211075,"score_gpt":0.3699668967910038,"score_spread":0.2521891552698963,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}