{"id":"W7082289080","doi":"10.48448/ffmv-8j70","title":"MAmmoTH-VL: Eliciting Multimodal Reasoning with Instruction Tuning at Scale","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Construct (python library); Scalability; Scale (ratio); Range (aeronautics); Cover (algebra)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0006008559,0.0003543723,0.0003194702,0.0004869917,0.0006698887,0.0002842052,0.001710935,0.0002229998,0.000198708],"category_scores_gemma":[0.0002076297,0.0003071554,0.00004869108,0.001433911,0.0006475088,0.0003386132,0.001091355,0.0003956179,0.00004819463],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002122192,"about_ca_system_score_gemma":0.000574487,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004360237,"about_ca_topic_score_gemma":0.0002736568,"domain_scores_codex":[0.9971406,0.00003939447,0.0002828497,0.001175033,0.0006779153,0.0006841354],"domain_scores_gemma":[0.9983175,0.00005980333,0.0003591847,0.0008784804,0.0002213157,0.0001637272],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007225604,0.0003714189,0.01862882,0.001207045,0.0002502881,0.0003520715,0.002844284,0.0119692,0.0328587,0.09963679,0.0573616,0.7744475],"study_design_scores_gemma":[0.001664133,0.0001719662,0.0008907472,0.003213255,0.00005743587,0.0004535671,0.0005941593,0.6167844,0.01357538,0.002683122,0.3579085,0.002003269],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.001225293,0.00008788474,0.2170551,0.0007344396,0.000528978,0.0002128814,0.000005804745,0.0006887169,0.7794609],"genre_scores_gemma":[0.09878249,0.00002607676,0.3868616,0.0003584007,0.0004304004,0.00002296565,0.00002921447,0.00005434488,0.5134345],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.7724442,"threshold_uncertainty_score":0.9999381,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.007702649910607454,"score_gpt":0.2303360047589924,"score_spread":0.222633354848385,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}