{"id":"W3176889315","doi":"","title":"Impact of Tokenization, Pretraining Task, and Transformer Depth on Text Ranking","year":2021,"lang":"en","type":"article","venue":"UvA-DARE (University of Amsterdam)","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Nederlandse Organisatie voor Wetenschappelijk Onderzoek; Canadian Institute of Steel Construction","keywords":"Computer science; Transformer; Lexical analysis; Preprocessor; Artificial intelligence; Natural language processing; Vocabulary; Machine learning; Question answering; Information retrieval; Linguistics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001175633,0.00009734728,0.0002024857,0.0001202872,0.0001098916,0.00002511723,0.0003146536,0.00005911651,0.00006537108],"category_scores_gemma":[0.00001862558,0.0001120822,0.0000986682,0.0003057267,0.00005360357,0.0004527158,0.00009264027,0.00009385338,0.00000239],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004125846,"about_ca_system_score_gemma":0.0001187428,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001182163,"about_ca_topic_score_gemma":0.00008184473,"domain_scores_codex":[0.9991892,0.00004958267,0.0001190194,0.0002739154,0.000210413,0.0001578883],"domain_scores_gemma":[0.9993303,0.00005801816,0.0000935983,0.0003015128,0.0001512314,0.00006532399],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0001391225,0.0002719596,0.0737195,0.0003406212,0.0003901684,0.0001686187,0.08703183,0.008265913,0.02111177,0.01633174,0.0004656815,0.7917631],"study_design_scores_gemma":[0.007873803,0.001371529,0.5906323,0.001245243,0.0001800746,0.0001570673,0.009295726,0.3751614,0.008385848,0.002624997,0.001729603,0.001342458],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5428854,0.00005245905,0.4529753,0.0002045976,0.00004665843,0.00005676175,0.000006382865,0.00002522876,0.003747209],"genre_scores_gemma":[0.9778236,0.00002147754,0.02192065,0.00003621192,0.00001170547,4.903997e-8,0.00000498621,0.000005050697,0.0001762383],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7904206,"threshold_uncertainty_score":0.4570581,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0165175014853993,"score_gpt":0.2246721207000233,"score_spread":0.208154619214624,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}