{"id":"W4416617284","doi":"10.1109/icspis68676.2025.11551719","title":"Enhancing Reasoning Skills in Small Persian Medical Language Models Can Outperform Large-Scale Data Training","year":2025,"lang":"","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Windsor","funders":"","keywords":"Persian; Baseline (sea); Language model; Training set; Preference; Verbal reasoning; Training (meteorology); Variety (cybernetics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002572019,0.001434838,0.0005536394,0.0004478876,0.0003544223,0.000899947,0.001499135,0.001042587,0.005554588],"category_scores_gemma":[0.01013316,0.0004078044,0.0008954548,0.0003520079,0.0006736138,0.001770786,0.001460158,0.002769822,0.002538916],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008195477,"about_ca_system_score_gemma":0.001913463,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004959032,"about_ca_topic_score_gemma":0.01201898,"domain_scores_codex":[0.9988943,0.0004345309,0.00007658881,0.0003428288,0.0001610599,0.00009069552],"domain_scores_gemma":[0.9961456,0.002657462,0.0001434387,0.0005195725,0.0003868384,0.0001469974],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001277145,0.0008120181,0.01235246,0.0008035625,0.0002283828,0.0004006439,0.0005051112,0.3568689,0.02611279,0.0039808,0.02256594,0.5740923],"study_design_scores_gemma":[0.0001424231,0.0003433013,0.001922894,0.0000648526,0.00006266241,0.0001172725,0.0001741225,0.9690807,0.01547783,0.005682075,0.006902523,0.00002924284],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4667846,0.003500459,0.4619948,0.003937759,0.0006485497,0.0005406323,0.002801382,0.03968247,0.02010928],"genre_scores_gemma":[0.7995741,0.0003813988,0.1873668,0.001452141,0.00007866839,0.0001989617,0.005088282,0.0005213171,0.005338513],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005554588,"threshold_uncertainty_score":0.01858193,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03256544562111078,"score_gpt":0.2876779793084234,"score_spread":0.2551125336873126,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}