{"id":"W4313589266","doi":"10.21203/rs.3.rs-2410184/v1","title":"Biomedical Text Readability and Cognitive Burden after Hypernym Substitution with Fine-Tuned Large Language Models","year":2023,"lang":"en","type":"preprint","venue":"Research Square","topic":"Text Readability and Simplification","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Readability; Substitution (logic); Computer science; Cognition; Natural language processing; Linguistics; Artificial intelligence; Psychology; Programming language; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004610658,0.0006200378,0.0008946463,0.001207978,0.0004672298,0.002104565,0.001202062,0.0009876917,0.00568975],"category_scores_gemma":[0.06330209,0.0004020785,0.0007544952,0.001274426,0.0009457447,0.004497441,0.001754269,0.001639893,0.00112509],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006498533,"about_ca_system_score_gemma":0.0009477301,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003498998,"about_ca_topic_score_gemma":0.003733641,"domain_scores_codex":[0.9964767,0.001740539,0.0003131481,0.0007021275,0.0005474812,0.0002199524],"domain_scores_gemma":[0.9026606,0.08153827,0.003046864,0.009279246,0.002668346,0.0008067491],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01425648,0.001785504,0.04342115,0.001348288,0.001126382,0.001944599,0.002948242,0.1497169,0.08307442,0.01226113,0.01399499,0.6741219],"study_design_scores_gemma":[0.0003744376,0.0007427757,0.02893523,0.0001076976,0.0006167901,0.0008271568,0.0009548718,0.8720025,0.04506361,0.04746808,0.002761684,0.0001451451],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8639942,0.0007755143,0.1263959,0.001098768,0.0001468116,0.00009251402,0.001538203,0.003057992,0.002900125],"genre_scores_gemma":[0.9633983,0.0001633512,0.03282064,0.00008010834,0.0000507924,0.00005000323,0.001675146,0.0004446868,0.001317058],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.00568975,"threshold_uncertainty_score":0.02438378,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0669732686738143,"score_gpt":0.3703418086020396,"score_spread":0.3033685399282253,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}