{"id":"W7126435203","doi":"10.21428/594757db.ba120f9f","title":"Same File Prediction: A New Pretraining Objective forBERT-like Transformers","year":2024,"lang":"en","type":"article","venue":"","topic":"Imbalanced Data Classification Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université Laval","funders":"","keywords":"Transformer; Language model; Training set; Task (project management); Task analysis; Binary classification","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0001571447,0.0001324615,0.0001097219,0.000157414,0.00007953584,0.0003060345,0.0005153625,0.00007951949,0.001505553],"category_scores_gemma":[0.00003036375,0.0001166902,0.00007202339,0.0006630042,0.00003706004,0.001451673,0.00006074728,0.0001833747,0.0001570733],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001026718,"about_ca_system_score_gemma":0.0002695378,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005447732,"about_ca_topic_score_gemma":0.00001287939,"domain_scores_codex":[0.9987918,0.00002197124,0.0002116801,0.0004809718,0.0002472539,0.0002463605],"domain_scores_gemma":[0.9993002,0.0001355547,0.000024281,0.0003940067,0.0000391424,0.0001067507],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000003466931,0.00001574372,0.00001066873,0.00002562617,0.00003556806,0.000007217398,0.002802568,0.000004852533,0.0009410827,0.08370139,0.5667837,0.3456681],"study_design_scores_gemma":[0.000242348,0.000263097,0.001072226,0.0001838897,0.00001519832,0.00008906938,0.0002418746,0.3214849,0.01202578,0.01909238,0.6448784,0.0004108118],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00001292239,0.0001355695,0.9412555,0.001589826,0.0004753165,0.0002360393,0.0001646544,0.002752592,0.05337762],"genre_scores_gemma":[0.1101161,0.00009466634,0.8452307,0.00304665,0.0004328846,0.0003423333,0.0003444465,0.00005443371,0.04033779],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.3452573,"threshold_uncertainty_score":0.9994072,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01981406480209573,"score_gpt":0.2594525832086411,"score_spread":0.2396385184065454,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}