{"id":"W6891769059","doi":"10.48448/x2c7-r841","title":"An Empirical Study of Multilingual Reasoning Distillation for Question Answering","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Correctness; Distillation; Variety (cybernetics); Empirical research; Question answering; Case-based reasoning; Empirical evidence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01337759,0.001238585,0.0007044513,0.00115587,0.001026865,0.002026727,0.001848723,0.001033581,0.007430785],"category_scores_gemma":[0.09368114,0.0005126284,0.0007009372,0.001577913,0.001598234,0.005639058,0.004395031,0.003928014,0.0029652],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001105211,"about_ca_system_score_gemma":0.001461871,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006845097,"about_ca_topic_score_gemma":0.008062094,"domain_scores_codex":[0.9849723,0.009875559,0.0009097648,0.002206039,0.001716432,0.0003199177],"domain_scores_gemma":[0.870562,0.1086434,0.003507851,0.01087846,0.004706951,0.001701396],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.006779686,0.004528493,0.1878386,0.003936904,0.0009898256,0.0009629271,0.006637921,0.02988636,0.01013037,0.01062343,0.0322208,0.7054648],"study_design_scores_gemma":[0.002626679,0.007697783,0.222199,0.001346685,0.001562407,0.004391211,0.01169439,0.4887644,0.04255823,0.04494719,0.1716111,0.0006009292],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9321544,0.004481649,0.03706906,0.001336045,0.0002709718,0.0003783962,0.004040497,0.002384514,0.01788453],"genre_scores_gemma":[0.9640566,0.0005864736,0.02376172,0.000340818,0.00007535869,0.0002505207,0.00768038,0.0003824241,0.002865722],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01337759,"threshold_uncertainty_score":0.07074833,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0436910017129451,"score_gpt":0.4279754972270292,"score_spread":0.3842844955140841,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}