{"id":"W4406040593","doi":"10.62051/h08exg91","title":"Comparative Evaluation of GPT, BERT, and XLNet: Insights into Their Performance and Applicability in NLP Tasks","year":2024,"lang":"en","type":"article","venue":"Transactions on Computer Science and Intelligent Systems Research","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Transformer; Language model; Artificial intelligence; Natural language processing; Natural language understanding; Encoder; Machine learning; Comprehension; Generative grammar; Natural language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003987227,0.002250684,0.0009669234,0.001198351,0.0004355957,0.001312454,0.002141664,0.001873759,0.002666987],"category_scores_gemma":[0.01413357,0.000511167,0.0007357445,0.0009076362,0.0008342298,0.003967865,0.001720175,0.003145908,0.001261056],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001773029,"about_ca_system_score_gemma":0.00171095,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01268898,"about_ca_topic_score_gemma":0.01973357,"domain_scores_codex":[0.998495,0.0006419175,0.00009945356,0.0003859629,0.0002381766,0.0001393966],"domain_scores_gemma":[0.9942909,0.004115596,0.0001878596,0.0005622194,0.0005839545,0.0002593264],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001412349,0.0007623179,0.01199661,0.000877062,0.0004019731,0.0002694023,0.0004489786,0.5429724,0.003945759,0.005343453,0.01486136,0.4167084],"study_design_scores_gemma":[0.00006926663,0.0006010209,0.002195559,0.00009073998,0.0000969796,0.00008985876,0.0001843112,0.9860373,0.004367119,0.003747972,0.002482601,0.00003731512],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6901668,0.01102848,0.241467,0.004016522,0.001040467,0.0008153098,0.004331314,0.0197615,0.02737266],"genre_scores_gemma":[0.8983963,0.002243533,0.08385669,0.0006796234,0.0001087175,0.0005223411,0.007488951,0.0005815384,0.006122295],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01268898,"threshold_uncertainty_score":0.02523029,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1690005470928217,"score_gpt":0.3991781544185883,"score_spread":0.2301776073257666,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}