{"id":"W2579578061","doi":"10.63317/2sfruoxcckcu","title":"Classifying Out-of-vocabulary Terms in a Domain-Specific Social Media Corpus","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of New Brunswick; University of Toronto","funders":"","keywords":"Vocabulary; Computer science; Natural language processing; Task (project management); Artificial intelligence; Domain (mathematical analysis); Word (group theory); Set (abstract data type); Linguistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008239347,0.0006234801,0.0004995127,0.007276343,0.001277178,0.001176247,0.0006358626,0.001033486,0.001500348],"category_scores_gemma":[0.006190927,0.0002459597,0.0006486403,0.004537428,0.000721351,0.001696676,0.0009482535,0.0009873873,0.0009455546],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000676089,"about_ca_system_score_gemma":0.001296041,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01008017,"about_ca_topic_score_gemma":0.02251799,"domain_scores_codex":[0.9987475,0.0002724691,0.0001901051,0.0002831724,0.0004120408,0.00009464736],"domain_scores_gemma":[0.9943116,0.003874972,0.0003607078,0.0003829953,0.000869569,0.0002001745],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001872531,0.001760933,0.1193732,0.005987329,0.0007653621,0.006256979,0.005456656,0.009393355,0.3042964,0.01026011,0.05348818,0.481089],"study_design_scores_gemma":[0.0003358063,0.0009377219,0.4329787,0.001231203,0.001415032,0.01279085,0.0135293,0.2113894,0.1284644,0.01040901,0.186229,0.0002897081],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9232972,0.005160824,0.03197735,0.0008539199,0.0004264478,0.0004357532,0.02978751,0.001141383,0.006919691],"genre_scores_gemma":[0.8638363,0.002253706,0.04613093,0.0002596178,0.00031587,0.0004003185,0.08296466,0.0002764615,0.003562183],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01008017,"threshold_uncertainty_score":0.02004296,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03021188006224321,"score_gpt":0.2698445468050732,"score_spread":0.23963266674283,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}