{"id":"W4214859661","doi":"10.31234/osf.io/m397u","title":"Concreteness ratings for 62 thousand English multiword expressions","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Concreteness; Psychology; Meaning (existential); Linguistics; Computer science; Word (group theory); Cognitive psychology; Natural language processing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0006230304,0.0003693836,0.0004398682,0.0001714771,0.0003967454,0.0006105308,0.003381171,0.0003013683,0.0001333451],"category_scores_gemma":[0.0007216854,0.0003135493,0.0002180442,0.0002183851,0.00005351434,0.0003514752,0.006635914,0.000960828,0.000001330437],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000104951,"about_ca_system_score_gemma":0.0002854812,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00008028114,"about_ca_topic_score_gemma":0.000004421087,"domain_scores_codex":[0.9975655,0.0001225039,0.0004112303,0.001072544,0.0004167131,0.0004114744],"domain_scores_gemma":[0.997335,0.0004788505,0.0003415094,0.001319296,0.0004121485,0.0001131576],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001884489,0.0003873544,0.0006207397,0.003409209,0.000352399,0.0001887236,0.03761704,0.0007893674,0.039962,0.507789,0.186758,0.2219378],"study_design_scores_gemma":[0.002277715,0.0004078944,0.00004714843,0.001380112,0.0001286502,0.000029341,0.0008637459,0.1536227,0.265488,0.3658506,0.2052708,0.004633307],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001105818,0.002231797,0.9871919,0.0004849598,0.001819327,0.001210255,0.0000615726,0.003588509,0.002305837],"genre_scores_gemma":[0.05366027,0.00002498457,0.9417539,0.000672155,0.0003012326,0.001352134,0.00007315653,0.00004352311,0.002118662],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.225526,"threshold_uncertainty_score":0.9999316,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0208887967583278,"score_gpt":0.3085123247574185,"score_spread":0.2876235279990907,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}