{"id":"W7133311834","doi":"10.70764/gdpu-bit.2025.1(2)-01","title":"Enhancing Zero-Shot Reasoning in Language Models Via Hybrid Instruction Marginalization","year":2025,"lang":"","type":"article","venue":"Breakthroughs Information Technology","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Component (thermodynamics); Selection (genetic algorithm); Argumentative; Cognition; Language model; Reasoning system; Isolation (microbiology); Verbal reasoning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00281769,0.001328854,0.0008519727,0.001129318,0.0005757441,0.002242093,0.002900593,0.0009206898,0.004621171],"category_scores_gemma":[0.01950461,0.0005017115,0.001192625,0.0005700285,0.001432507,0.004091606,0.00375687,0.001749867,0.001114904],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009597075,"about_ca_system_score_gemma":0.002166545,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002241147,"about_ca_topic_score_gemma":0.004101136,"domain_scores_codex":[0.9976291,0.0008768262,0.0001423501,0.0006465884,0.000562629,0.0001425535],"domain_scores_gemma":[0.992781,0.004464298,0.0004293431,0.001428771,0.0007126941,0.0001839319],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.000755349,0.0005977201,0.00993167,0.0008709839,0.0001869281,0.0003809379,0.002397886,0.1172443,0.0391527,0.03015667,0.004407376,0.7939175],"study_design_scores_gemma":[0.00007754692,0.0002740487,0.001521727,0.00009220904,0.0001130728,0.0002182955,0.0003372267,0.9015537,0.03416671,0.05404624,0.007521664,0.00007759072],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06854682,0.0003829157,0.9162569,0.0003134923,0.00003861065,0.0001325369,0.0002274654,0.01158996,0.002511203],"genre_scores_gemma":[0.4958444,0.0001208762,0.5008283,0.0002339987,0.00002643076,0.0001767704,0.0006938966,0.0008226141,0.001252781],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004621171,"threshold_uncertainty_score":0.01545936,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.008059412533625533,"score_gpt":0.243575538514773,"score_spread":0.2355161259811475,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}