{"id":"W4392400363","doi":"10.23977/jaip.2024.070110","title":"Exploration of Deep Learning Evaluation from the Perspective of Multimodal Data Analysis","year":2024,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Practice","topic":"Medical Research and Treatments","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Perspective (graphical); Computer science; Deep learning; Artificial intelligence; Machine learning; Natural language processing; Data science","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.003991563,0.000070095,0.0002526594,0.0002088111,0.00004795805,0.00004055123,0.0002006689,0.00004583129,0.0005514672],"category_scores_gemma":[0.03513413,0.00004229937,0.000133201,0.0007993034,0.0001076436,0.001050164,0.00005325069,0.0004827209,0.0000212483],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001126627,"about_ca_system_score_gemma":0.0003921981,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001197096,"about_ca_topic_score_gemma":0.0001368772,"domain_scores_codex":[0.9973042,0.0005037225,0.0006207997,0.0001632052,0.001306826,0.0001012266],"domain_scores_gemma":[0.9939293,0.003448476,0.0004861817,0.0002836728,0.001754945,0.00009740378],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00281074,0.0009808514,0.0008906407,0.00003422823,0.006616382,0.0001541609,0.0156974,0.004752538,0.007196837,0.002822658,0.0001184253,0.9579251],"study_design_scores_gemma":[0.0001707492,0.001262177,0.001320064,0.0002751316,0.008008423,0.00003118104,0.06128013,0.8916228,0.02142184,0.01361065,0.00092418,0.00007267202],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08934169,0.01133987,0.8797889,0.01742182,0.0004102316,0.0004506007,0.00002665759,0.000009913808,0.001210326],"genre_scores_gemma":[0.991919,0.001282636,0.006418243,0.00003701515,0.0002914174,0.000001966897,0.00003073227,0.000005982195,0.00001300339],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9578525,"threshold_uncertainty_score":0.9729934,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2850112876842015,"score_gpt":0.5199762326703001,"score_spread":0.2349649449860986,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}