{"id":"W4396883611","doi":"10.31234/osf.io/uc6d4","title":"Prompting sometimes invokes expert-like downward shifts in multimodal models’ conceptual hierarchies","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Speech and dialogue systems","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Psychology; Cognitive psychology; Epistemology; Cognitive science; Philosophy","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0006325644,0.0006071258,0.0007730863,0.000772235,0.0000886257,0.0008684107,0.00207978,0.0004713463,0.00002871137],"category_scores_gemma":[0.00007162729,0.0005102271,0.0002762843,0.0004621168,0.0002129221,0.0004597544,0.005503033,0.001142298,0.0002316506],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001751344,"about_ca_system_score_gemma":0.0005928004,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003519661,"about_ca_topic_score_gemma":0.0004973875,"domain_scores_codex":[0.9958703,0.00023983,0.0008209205,0.001593719,0.0006932814,0.0007819674],"domain_scores_gemma":[0.9981394,0.0001933674,0.0001671958,0.001180129,0.0000913385,0.0002286018],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001360706,0.0007718616,0.001040478,0.001961401,0.0004651323,0.001037011,0.2981399,0.06511719,0.002112333,0.4597503,0.03129579,0.1381725],"study_design_scores_gemma":[0.0007241547,0.00009562912,0.0002067192,0.0009460063,0.0000103292,0.00001384456,0.002034315,0.7891128,0.0009962681,0.2032422,0.001338095,0.001279666],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4578862,0.03251308,0.3713676,0.00456654,0.02306112,0.005542938,0.0001450457,0.006246652,0.09867076],"genre_scores_gemma":[0.8950285,0.00008177729,0.1024959,0.0003499671,0.0005735953,0.0004109128,0.00003746734,0.00004773458,0.0009741723],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7239956,"threshold_uncertainty_score":0.9997349,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04430763271231802,"score_gpt":0.2735097836492447,"score_spread":0.2292021509369266,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}