{"id":"W2610765827","doi":"10.1515/phras-2013-0003","title":"In support of multiword unit classifications: Corpus and human rating data validate phraseological classifications of three different multiword unit types","year":2013,"lang":"en","type":"article","venue":"Yearbook of Phraseology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Computer science; Natural language processing; Linguistics; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01344518,0.0003969788,0.0004528923,0.002108776,0.00124787,0.003495291,0.0009249555,0.0008321175,0.004108733],"category_scores_gemma":[0.09435833,0.0003019713,0.0002503834,0.002462948,0.002905505,0.003994831,0.002386595,0.00121851,0.001212037],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004713197,"about_ca_system_score_gemma":0.0003582647,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002978282,"about_ca_topic_score_gemma":0.004419092,"domain_scores_codex":[0.9876392,0.006457864,0.001471787,0.001958446,0.002254228,0.0002183971],"domain_scores_gemma":[0.8487098,0.09178989,0.01568273,0.02274063,0.01988507,0.001191962],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.002262805,0.0007001891,0.5300153,0.002032371,0.000334473,0.0006348121,0.1445934,0.00111031,0.04789374,0.01555612,0.008370682,0.2464958],"study_design_scores_gemma":[0.0001424795,0.000745156,0.9067391,0.000634191,0.0001446699,0.001565057,0.03977481,0.00856002,0.01192334,0.01005009,0.01951432,0.0002067245],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9635739,0.0008414236,0.01379121,0.0002845127,0.00009307463,0.0002402789,0.0007206654,0.00006223092,0.02039273],"genre_scores_gemma":[0.9884492,0.0002027471,0.008964125,0.0001350305,0.00003765149,0.0003343931,0.0009096415,0.00005907276,0.0009080861],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01344518,"threshold_uncertainty_score":0.07110578,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1005506097616091,"score_gpt":0.3419018279530505,"score_spread":0.2413512181914414,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}