{"id":"W3211439810","doi":"10.18653/v1/2021.emnlp-main.835","title":"Types of Out-of-Distribution Texts and How to Detect Them","year":2021,"lang":"en","type":"article","venue":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"York University","keywords":"Computer science; Categorization; Artificial intelligence; Calibration; Natural language processing; Data mining; Statistics; Mathematics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005784863,0.0008816297,0.0009364073,0.006984505,0.001090183,0.004401062,0.001602334,0.002203989,0.002861118],"category_scores_gemma":[0.07817785,0.00055587,0.0007307596,0.003228913,0.001324851,0.007493149,0.002227371,0.001968752,0.002795136],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007571501,"about_ca_system_score_gemma":0.0007216017,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001408438,"about_ca_topic_score_gemma":0.002275611,"domain_scores_codex":[0.9901248,0.003549391,0.001108926,0.002215523,0.002642371,0.000358915],"domain_scores_gemma":[0.938959,0.04427333,0.005402951,0.005192767,0.005186823,0.0009851063],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001073732,0.0003381698,0.3041928,0.004581575,0.000417866,0.002502373,0.008854931,0.007849592,0.02504239,0.02395971,0.06861564,0.5525713],"study_design_scores_gemma":[0.000212204,0.0003581676,0.2458381,0.002462381,0.0004628376,0.01605955,0.01956075,0.3659904,0.07120726,0.09008551,0.1872862,0.0004766865],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6693535,0.008470272,0.269127,0.007282251,0.001058917,0.0008162445,0.01598664,0.007240902,0.02066438],"genre_scores_gemma":[0.8251305,0.001683383,0.1519521,0.000688033,0.0003925987,0.0003822305,0.01446866,0.0009002696,0.004402319],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006984505,"threshold_uncertainty_score":0.03059369,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0736749441821645,"score_gpt":0.3899331424867225,"score_spread":0.316258198304558,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}