{"id":"W4405702407","doi":"10.54660/.ijmrge.2024.5.6.1279-1286","title":"Analyzing and Mitigating Dataset Artifacts in Natural Language Inference Models Using ELECTRA","year":2024,"lang":"en","type":"article","venue":"International Journal of Multidisciplinary Research and Growth Evaluation","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Inference; Computer science; Artificial intelligence; Robustness (evolution); Weighting; Regularization (linguistics); Adversarial system; Generalization; Natural language understanding; Artifact (error); Machine learning; Natural language processing; Natural language; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003913943,0.00008046004,0.0001067243,0.0007355331,0.00008558087,0.0004452828,0.0003877181,0.00003637294,0.000003796827],"category_scores_gemma":[0.0004709874,0.00006915628,0.00002189452,0.0003008628,0.00005341109,0.002058976,0.0003500909,0.0004746678,8.178607e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001738304,"about_ca_system_score_gemma":0.0002664187,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00009395085,"about_ca_topic_score_gemma":0.00003649838,"domain_scores_codex":[0.9979218,0.0002355312,0.0003600849,0.0002445374,0.001038686,0.0001994016],"domain_scores_gemma":[0.99865,0.0004609888,0.00008075566,0.00009925489,0.0006277584,0.00008130606],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000288807,0.0002973249,0.03405906,0.0002857135,0.00035524,0.001562462,0.02722932,0.04506514,0.140138,0.04696633,0.0001271522,0.7036254],"study_design_scores_gemma":[0.000313205,0.00006283438,0.004628106,0.0003368404,0.000004895898,0.0001416702,0.0002223934,0.9608194,0.0009126276,0.03249172,0.000001159285,0.00006513957],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8908655,0.00367094,0.1039216,0.001168989,0.0002236263,0.000106223,0.000009966604,0.000007586248,0.00002555865],"genre_scores_gemma":[0.9846334,0.0001522501,0.01502622,0.000008316111,0.0001463422,0.000002855555,0.00002087269,0.000004826791,0.000004931912],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9157543,"threshold_uncertainty_score":0.4293873,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1479984930523851,"score_gpt":0.4685742009250235,"score_spread":0.3205757078726384,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}