{"id":"W2124602574","doi":"10.1145/502512.502558","title":"Induction of semantic classes from natural language text","year":2001,"lang":"en","type":"article","venue":"","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":99,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Natural language processing; Artificial intelligence; Set (abstract data type); Task (project management); Space (punctuation); Natural language; Unsupervised learning; Information retrieval","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001672562,0.0009821595,0.000749794,0.006785381,0.001742534,0.001380293,0.001352117,0.0008871856,0.002952703],"category_scores_gemma":[0.008946538,0.0004115948,0.001278647,0.003122626,0.001352201,0.003660142,0.001872757,0.001886307,0.001849902],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001204149,"about_ca_system_score_gemma":0.002346931,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001993676,"about_ca_topic_score_gemma":0.004000432,"domain_scores_codex":[0.997725,0.0005281327,0.0002031275,0.0007042468,0.0006990386,0.0001405317],"domain_scores_gemma":[0.9941854,0.003513986,0.0004864013,0.0005615124,0.001091661,0.0001610724],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002948279,0.0004482083,0.006577735,0.0006512877,0.00008358462,0.0002845388,0.001029004,0.006236929,0.01668105,0.03977891,0.02406508,0.9038689],"study_design_scores_gemma":[0.00016887,0.0002520733,0.01478467,0.0004791487,0.000206338,0.0009713718,0.00219603,0.4643527,0.06131,0.3263514,0.1287586,0.000168882],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05537136,0.0008557125,0.92278,0.001192693,0.0002773623,0.0008198968,0.004421255,0.005594963,0.008686723],"genre_scores_gemma":[0.1445165,0.0005455428,0.8332618,0.0002849956,0.0002250358,0.0008640752,0.01678769,0.0004227733,0.003091658],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006785381,"threshold_uncertainty_score":0.009877741,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01160689930725286,"score_gpt":0.2518886837407619,"score_spread":0.240281784433509,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}