{"id":"W4416417137","doi":"10.1007/s11432-025-4676-4","title":"Large multimodal models evaluation: a survey","year":2025,"lang":"en","type":"article","venue":"Science China Information Sciences","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Competence (human resources); Data collection; Evaluation methods; Term (time); Multimodal interaction","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01094652,0.003199078,0.004004644,0.004708426,0.0008721894,0.002715803,0.00462891,0.001915481,0.009576092],"category_scores_gemma":[0.03478954,0.0007605539,0.001882047,0.003876067,0.0007996369,0.005115622,0.002763778,0.001438139,0.002317463],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001761062,"about_ca_system_score_gemma":0.002085237,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007573685,"about_ca_topic_score_gemma":0.008610157,"domain_scores_codex":[0.9920855,0.003736737,0.0005829742,0.001056976,0.002312757,0.0002249575],"domain_scores_gemma":[0.9831362,0.01235098,0.0003938426,0.001644586,0.002176396,0.0002980556],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0005458355,0.0004119263,0.004580502,0.002143637,0.0005432622,0.00007450096,0.0000771598,0.0249511,0.001026168,0.003227334,0.03491523,0.9275033],"study_design_scores_gemma":[0.0003459248,0.001904027,0.01599281,0.003087522,0.002035683,0.001097044,0.0008756125,0.7972913,0.01043488,0.04172902,0.1250182,0.0001879564],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.0724339,0.5040843,0.3700435,0.005699029,0.001331082,0.0007566923,0.00544887,0.007571402,0.03263121],"genre_scores_gemma":[0.574582,0.1389612,0.2363102,0.002880557,0.001954691,0.00107881,0.02334801,0.003308487,0.01757593],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.9890535,"threshold_uncertainty_score":0.05789137,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03383829650789999,"score_gpt":0.3639713949218057,"score_spread":0.3301330984139058,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}