{"id":"W7126435203","doi":"10.21428/594757db.ba120f9f","title":"Same File Prediction: A New Pretraining Objective forBERT-like Transformers","year":2024,"lang":"en","type":"article","venue":"","topic":"Imbalanced Data Classification Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université Laval","funders":"","keywords":"Transformer; Language model; Training set; Task (project management); Task analysis; Binary classification","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00138889,0.001295212,0.0006955834,0.0007986169,0.0005219066,0.0008426335,0.001734584,0.001228691,0.00303738],"category_scores_gemma":[0.004288088,0.0005161419,0.0006552619,0.0006432247,0.0006195419,0.002511937,0.001501668,0.002654325,0.001721198],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007785841,"about_ca_system_score_gemma":0.001414811,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006324485,"about_ca_topic_score_gemma":0.01382538,"domain_scores_codex":[0.9994266,0.000138818,0.00003649171,0.000210308,0.0001041677,0.00008370493],"domain_scores_gemma":[0.9984364,0.0007888524,0.0001018554,0.0002415811,0.0003450518,0.00008629051],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0005991231,0.0005990608,0.01019848,0.0001075688,0.0000945693,0.0002108098,0.0002036885,0.2602116,0.03259106,0.005680296,0.01068816,0.6788156],"study_design_scores_gemma":[0.00001101492,0.0000766867,0.0006533215,0.000006874993,0.00001275401,0.00004150634,0.00002537116,0.9903916,0.006498188,0.001308866,0.000964262,0.000009524332],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1246053,0.0003724098,0.8638257,0.0007609306,0.0001487078,0.0001842698,0.0004664324,0.006275439,0.003360808],"genre_scores_gemma":[0.770503,0.0002164129,0.2162561,0.0006198251,0.0001345897,0.0002651395,0.002063295,0.0004507195,0.009490915],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006324485,"threshold_uncertainty_score":0.01257533,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01981406480209573,"score_gpt":0.2594525832086411,"score_spread":0.2396385184065454,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}