{"id":"W3176889315","doi":"","title":"Impact of Tokenization, Pretraining Task, and Transformer Depth on Text Ranking","year":2021,"lang":"en","type":"article","venue":"UvA-DARE (University of Amsterdam)","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Nederlandse Organisatie voor Wetenschappelijk Onderzoek; Canadian Institute of Steel Construction","keywords":"Computer science; Transformer; Lexical analysis; Preprocessor; Artificial intelligence; Natural language processing; Vocabulary; Machine learning; Question answering; Information retrieval; Linguistics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004420233,0.001502977,0.001049603,0.0006566606,0.0007462113,0.002145054,0.001658259,0.001582126,0.006754209],"category_scores_gemma":[0.03161162,0.0005209697,0.0006603738,0.0009030909,0.001074843,0.008130583,0.00241783,0.002686282,0.003685436],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001028905,"about_ca_system_score_gemma":0.002446692,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005941891,"about_ca_topic_score_gemma":0.007069738,"domain_scores_codex":[0.9978167,0.0008734432,0.0001881689,0.0004739793,0.000292603,0.0003551309],"domain_scores_gemma":[0.9873714,0.009069157,0.0004521155,0.001624404,0.000863521,0.0006194048],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.007212116,0.001438635,0.01098471,0.001101841,0.0001974003,0.0003508655,0.0004398813,0.09818333,0.04724128,0.005710559,0.01898864,0.8081508],"study_design_scores_gemma":[0.001084043,0.00680234,0.01297573,0.0003528326,0.0007677214,0.0008482089,0.001414458,0.8250841,0.1075204,0.02337705,0.01954929,0.0002236941],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7418754,0.007667572,0.2001847,0.002870327,0.001197296,0.0006711889,0.001831122,0.0243644,0.01933802],"genre_scores_gemma":[0.9172028,0.001139978,0.06683291,0.0009470481,0.0002102505,0.0002069263,0.003420528,0.001654911,0.008384642],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006754209,"threshold_uncertainty_score":0.0233767,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0165175014853993,"score_gpt":0.2246721207000233,"score_spread":0.208154619214624,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}