{"id":"W2923890923","doi":"10.48550/arxiv.1903.10972","title":"Models and Data for Simple Applications of BERT for Ad Hoc Document Retrieval","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":133,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Simple (philosophy); Sentence; Information retrieval; Microblogging; Inference; Question answering; Social media; Post hoc; Artificial intelligence; Natural language processing; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004184459,0.0009097279,0.0007430246,0.001873848,0.0007886645,0.00195519,0.002555566,0.001723306,0.00855207],"category_scores_gemma":[0.01878857,0.0006361115,0.0009267079,0.002249957,0.0007200402,0.004319523,0.001087251,0.00247604,0.003751051],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001992582,"about_ca_system_score_gemma":0.001195961,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01127221,"about_ca_topic_score_gemma":0.02126694,"domain_scores_codex":[0.9987115,0.0005011586,0.0001044536,0.0002761238,0.0003117895,0.0000949815],"domain_scores_gemma":[0.9931538,0.004322365,0.0004321149,0.001276149,0.0006779463,0.0001376907],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000759701,0.0004163289,0.01145227,0.0005800502,0.0001588371,0.0003897256,0.0005186087,0.5145147,0.005382609,0.20425,0.04310464,0.2184725],"study_design_scores_gemma":[0.0000268339,0.00004079983,0.001314471,0.00002706252,0.00001412395,0.000123142,0.00004289967,0.9096616,0.001078837,0.07751178,0.01012951,0.0000288566],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04475464,0.001162663,0.9284828,0.002768616,0.000178243,0.000351268,0.007656042,0.004924961,0.009720769],"genre_scores_gemma":[0.5416336,0.001171716,0.4225236,0.0006687716,0.0005918376,0.001135661,0.02071628,0.0005823597,0.01097627],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01127221,"threshold_uncertainty_score":0.02860951,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1510945483297511,"score_gpt":0.2488475987278627,"score_spread":0.09775305039811166,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}