{"id":"W4380355772","doi":"10.1613/jair.1.14157","title":"Your Prompt is My Command: On Assessing the Human-Centred Generality of Multimodal Models","year":2023,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Agencia Estatal de Investigación; Directorate-General for Communications Networks, Content and Technology; European Regional Development Fund; Generalitat Valenciana; Social Sciences and Humanities Research Council of Canada; Defense Advanced Research Projects Agency; European Commission; Universitat Politècnica de València; Future of Life Institute","keywords":"Generality; Computer science; Human–computer interaction; Cognition; Process (computing); Artificial intelligence; Cognitive science; Psychology; Programming language","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05152148,0.0007838203,0.0005711842,0.002331662,0.001234807,0.006615697,0.001196363,0.002039442,0.003152858],"category_scores_gemma":[0.2782211,0.0005112106,0.0007931726,0.001241457,0.006634658,0.01422398,0.006903527,0.002331261,0.0004400025],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002305099,"about_ca_system_score_gemma":0.0009678712,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002604756,"about_ca_topic_score_gemma":0.002520784,"domain_scores_codex":[0.9731517,0.01902322,0.001192623,0.002354914,0.003727504,0.0005500514],"domain_scores_gemma":[0.6783847,0.2810946,0.009728445,0.0189354,0.007948028,0.003908721],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.007466722,0.001280281,0.2383769,0.002958212,0.0009939843,0.0006033153,0.07660978,0.07175659,0.01773651,0.07008133,0.00521477,0.5069216],"study_design_scores_gemma":[0.0006285766,0.007821455,0.3218131,0.00235333,0.0009689876,0.002052374,0.06102499,0.3172678,0.02710017,0.2326368,0.02529314,0.001039283],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8465572,0.00178946,0.1140674,0.002567181,0.00009175559,0.0007340016,0.0002608679,0.0006396589,0.03329261],"genre_scores_gemma":[0.97643,0.0002805219,0.02234298,0.0001969215,0.0000220196,0.0001137728,0.0001157929,0.00006936031,0.0004285987],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.05152148,"threshold_uncertainty_score":0.2724749,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5477049452645087,"score_gpt":0.5082094541283876,"score_spread":0.03949549113612116,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}