{"id":"W4214859661","doi":"10.31234/osf.io/m397u","title":"Concreteness ratings for 62 thousand English multiword expressions","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Concreteness; Psychology; Meaning (existential); Linguistics; Computer science; Word (group theory); Cognitive psychology; Natural language processing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001977665,0.0006129485,0.0004378368,0.00190151,0.0004688272,0.0009111797,0.0002964793,0.0008045312,0.008764706],"category_scores_gemma":[0.02349244,0.0001775053,0.0004942215,0.001370844,0.0006127995,0.001475387,0.001218864,0.000750732,0.003528257],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003253449,"about_ca_system_score_gemma":0.0001959435,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008261506,"about_ca_topic_score_gemma":0.002108513,"domain_scores_codex":[0.9967934,0.0006746941,0.0007485202,0.000375999,0.001280635,0.0001268055],"domain_scores_gemma":[0.9812357,0.009911238,0.002155135,0.001357543,0.004464026,0.0008764],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.004557008,0.0009834219,0.2777147,0.0048071,0.0004977123,0.001622428,0.02266069,0.002409307,0.09760463,0.007111219,0.1193492,0.4606827],"study_design_scores_gemma":[0.0001536426,0.001378936,0.830207,0.0004719159,0.0002039093,0.002687979,0.009889739,0.007395147,0.02139333,0.004228571,0.121585,0.0004048558],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"dataset","genre_scores_codex":[0.961575,0.0009337291,0.005330218,0.0002591977,0.0002520601,0.000246269,0.01250107,0.0005385524,0.01836388],"genre_scores_gemma":[0.9292503,0.0007816899,0.01953747,0.0002532046,0.0001628667,0.001005676,0.03014186,0.0004790786,0.01838789],"genre_candidate":"dataset","genre_consensus":null,"teacher_disagreement_score":0.008764706,"threshold_uncertainty_score":0.0293209,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0208887967583278,"score_gpt":0.3085123247574185,"score_spread":0.2876235279990907,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}