{"id":"W4312738848","doi":"10.4000/books.aaccademia.11009","title":"Tackling Italian University Assessment Tests with Transformer-Based Language Models","year":2022,"lang":"en","type":"book-chapter","venue":"Accademia University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Canadian Institute for Advanced Research","keywords":"Transformer; Computer science; Security token; Natural language processing; Artificial intelligence; Task (project management); Cloze test; Test (biology); Mathematics education; Reading (process); Reading comprehension; Linguistics; Psychology; Engineering; Computer security; Systems engineering","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001059822,0.001281977,0.0005530455,0.0007279443,0.0002573836,0.001630369,0.001034075,0.0008113203,0.003216284],"category_scores_gemma":[0.003777227,0.0003996197,0.0007367967,0.0005849238,0.0003636007,0.001441119,0.001047716,0.001732647,0.002310284],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009567327,"about_ca_system_score_gemma":0.001392608,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01116878,"about_ca_topic_score_gemma":0.01358935,"domain_scores_codex":[0.9993746,0.0002548421,0.00003502675,0.0001823847,0.0001009977,0.00005211972],"domain_scores_gemma":[0.9984762,0.001139531,0.00007361676,0.0001139417,0.000149259,0.00004748814],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002273451,0.000171851,0.004215219,0.0001780646,0.00008288902,0.000337385,0.0003853771,0.2073827,0.009735116,0.00806857,0.007706812,0.7615086],"study_design_scores_gemma":[0.00001210689,0.00007090341,0.001045025,0.00001510639,0.00002353778,0.0001188391,0.0001150738,0.9843976,0.004585767,0.007148883,0.002449615,0.000017562],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1070857,0.0007145416,0.8714277,0.0004076498,0.00008743994,0.000157111,0.0004929501,0.009366491,0.01026046],"genre_scores_gemma":[0.6480911,0.0005345318,0.3352411,0.0001997177,0.0000688398,0.0001869861,0.002515687,0.0005978787,0.01256422],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01116878,"threshold_uncertainty_score":0.02220756,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0183020111558259,"score_gpt":0.2352954042576568,"score_spread":0.2169933931018309,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}