{"id":"W4402364429","doi":"10.1007/978-3-031-71167-1_2","title":"Assessing Logical Reasoning Capabilities of Encoder-Only Transformer Models","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Formal Methods in Verification","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Encoder; Transformer; Programming language; Electrical engineering; Operating system; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005744002,0.0007008617,0.0006425458,0.001065386,0.0004461757,0.00322984,0.002235318,0.001648896,0.007745976],"category_scores_gemma":[0.04183219,0.0006155848,0.001214918,0.0006531202,0.001496088,0.008475284,0.002525294,0.001601318,0.0009480524],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001177839,"about_ca_system_score_gemma":0.001641753,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002437914,"about_ca_topic_score_gemma":0.00306278,"domain_scores_codex":[0.9959274,0.001546437,0.0002205363,0.0005386567,0.001401999,0.0003649105],"domain_scores_gemma":[0.946393,0.0442967,0.001213932,0.005552283,0.001796775,0.0007473743],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004657588,0.001185784,0.01296055,0.0007699303,0.000424099,0.0005179297,0.0009024553,0.5771709,0.02163928,0.2304316,0.003476136,0.1458637],"study_design_scores_gemma":[0.0001110424,0.0002453361,0.0004182639,0.00003206039,0.0001274144,0.0001004503,0.0001495063,0.9096546,0.01291285,0.07542057,0.0008060709,0.00002187551],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5853966,0.0003909693,0.3911499,0.0007117653,0.00008360823,0.0001799413,0.001011284,0.00286354,0.01821245],"genre_scores_gemma":[0.9540342,0.0001179439,0.04374743,0.00004516063,0.00001490011,0.00002835215,0.0006293383,0.0001356958,0.001246877],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.007745976,"threshold_uncertainty_score":0.03037751,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05022977370404413,"score_gpt":0.3112825095173428,"score_spread":0.2610527358132986,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}