{"id":"W4312738848","doi":"10.4000/books.aaccademia.11009","title":"Tackling Italian University Assessment Tests with Transformer-Based Language Models","year":2022,"lang":"en","type":"book-chapter","venue":"Accademia University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Canadian Institute for Advanced Research","keywords":"Transformer; Computer science; Security token; Natural language processing; Artificial intelligence; Task (project management); Cloze test; Test (biology); Mathematics education; Reading (process); Reading comprehension; Linguistics; Psychology; Engineering; Computer security; Systems engineering","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0002173131,0.0005270392,0.0004659872,0.0004358354,0.0005170295,0.0001082716,0.002929103,0.0005210258,0.00003976677],"category_scores_gemma":[0.000003365768,0.0005734772,0.0002014186,0.00006597066,0.0001773161,0.0009271494,0.0006202744,0.001863745,0.000001516123],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008935441,"about_ca_system_score_gemma":0.000624202,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004206721,"about_ca_topic_score_gemma":0.0000461769,"domain_scores_codex":[0.9975707,0.00008685117,0.0001919489,0.0009558969,0.0007426993,0.0004519082],"domain_scores_gemma":[0.9982725,0.0001309654,0.0003411476,0.0009061737,0.0001399224,0.0002092328],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00007568526,0.00002020918,0.00000266998,0.0001132895,0.0001134176,0.001761044,0.000741629,0.0003182951,0.0001649696,0.9851651,0.0004702089,0.01105344],"study_design_scores_gemma":[0.00290713,0.0006123495,0.000004802463,0.0009697143,0.0006330133,0.000106824,0.0004829125,0.02030612,0.003838322,0.009320606,0.9576002,0.00321803],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.00006300514,0.0002946364,0.2518304,0.0001685394,0.00007025961,0.0005766045,0.0001363171,0.001280705,0.7455795],"genre_scores_gemma":[0.02572763,0.00008488778,0.2283598,0.0003373657,0.00006252545,0.000001999142,0.0001164614,0.0001107815,0.7451986],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.9758446,"threshold_uncertainty_score":0.9996716,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0183020111558259,"score_gpt":0.2352954042576568,"score_spread":0.2169933931018309,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}