{"id":"W4400526764","doi":"10.1145/3626772.3657824","title":"MTMS: Multi-teacher Multi-stage Knowledge Distillation for Reasoning-Based Machine Reading Comprehension","year":2024,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"Universitas Brawijaya","keywords":"Computer science; Distillation; Comprehension; Artificial intelligence; Reading (process); Reading comprehension; Natural language processing; Stage (stratigraphy); Programming language; Linguistics; Chemistry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003975506,0.0001874306,0.0001734914,0.0001508007,0.0001656837,0.0002483967,0.0003633997,0.00008801444,0.00003363781],"category_scores_gemma":[0.0000901309,0.0001589347,0.0001103065,0.0002397306,0.00002105367,0.0003338041,0.0001438426,0.0001523992,0.00006251669],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001134323,"about_ca_system_score_gemma":0.00007000762,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001066613,"about_ca_topic_score_gemma":0.0001129277,"domain_scores_codex":[0.998604,0.00006065203,0.0002834954,0.0006205859,0.00014555,0.0002857419],"domain_scores_gemma":[0.9990634,0.0002236874,0.00004556962,0.0004900926,0.00008312061,0.00009408656],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007168128,0.0009614837,0.00877364,0.001621187,0.0001608875,0.00007000919,0.007964073,0.09379331,0.02036011,0.2720821,0.003288962,0.5908526],"study_design_scores_gemma":[0.0005701667,0.00002774191,0.0006015204,0.0001406801,0.000008791941,0.000002256191,0.00001675643,0.9837374,0.0008571841,0.00007163369,0.01376003,0.0002058257],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.005777823,0.0007399203,0.9904627,0.0003163232,0.0007266208,0.0003900042,0.000007145727,0.0008813078,0.0006981863],"genre_scores_gemma":[0.5782616,0.00000212021,0.4163521,0.00005188851,0.00007557713,0.0000241362,0.00001660831,0.00001908191,0.005196876],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8899441,"threshold_uncertainty_score":0.648117,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08466576282071729,"score_gpt":0.3485421448935959,"score_spread":0.2638763820728786,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}