{"id":"W4405746722","doi":"10.1016/j.neunet.2024.107067","title":"Promises and perils of using Transformer-based models for SE research","year":2024,"lang":"en","type":"review","venue":"Neural Networks","topic":"Software Engineering Research","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Fundamental Research Funds for the Central Universities; Sun Yat-sen University","keywords":"Transformer; Computer science; Artificial intelligence; Machine learning; Engineering; Electrical engineering; Voltage","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.005721551,0.001302701,0.0009636818,0.002021746,0.0002420396,0.002294666,0.002441506,0.001544007,0.001938813],"category_scores_gemma":[0.01529046,0.0005553251,0.000979104,0.002130006,0.0008034621,0.006678873,0.001075502,0.003123194,0.001842508],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001534278,"about_ca_system_score_gemma":0.002369181,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004753241,"about_ca_topic_score_gemma":0.007700743,"domain_scores_codex":[0.9983085,0.0007542619,0.0001093402,0.0003235898,0.0004334266,0.00007079053],"domain_scores_gemma":[0.9886615,0.008325265,0.0002468571,0.000998574,0.001611681,0.0001560098],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001001602,0.0001077738,0.003224198,0.003112749,0.0003835575,0.00007821368,0.0001945143,0.06841533,0.00142771,0.05254054,0.01318423,0.8572311],"study_design_scores_gemma":[0.00005390933,0.0005453143,0.003425602,0.003820552,0.0005660729,0.0004793826,0.0003196437,0.5772849,0.005818314,0.2133777,0.1941611,0.0001475791],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.02087539,0.4625161,0.473913,0.01984614,0.00113668,0.0001717105,0.001108166,0.002082593,0.01835023],"genre_scores_gemma":[0.3105167,0.4437697,0.2267209,0.003749457,0.001239609,0.0004659898,0.003122422,0.0005154828,0.009899816],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.9942784,"threshold_uncertainty_score":0.03025883,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3153174685816854,"score_gpt":0.4532480130661218,"score_spread":0.1379305444844364,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}