{"id":"W107766491","doi":"10.1007/978-3-319-09912-5_1","title":"A Semi-supervised Learning Algorithm for Web Information Extraction with Tolerance Rough Sets","year":2014,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Rough Sets and Fuzzy Logic","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Winnipeg","funders":"","keywords":"Computer science; Categorical variable; Noun phrase; Artificial intelligence; Phrase; Natural language processing; Vector space model; Document classification; Machine learning; Information extraction; Naive Bayes classifier; Representation (politics); Data mining; Algorithm; Noun; Support vector machine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002022592,0.0006826128,0.002512908,0.002016116,0.001052644,0.001722479,0.002121201,0.001330354,0.002489693],"category_scores_gemma":[0.004221581,0.0007999974,0.002099072,0.002012004,0.0005730988,0.002122181,0.001762626,0.001497499,0.001291958],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006652219,"about_ca_system_score_gemma":0.001667868,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002460325,"about_ca_topic_score_gemma":0.002579817,"domain_scores_codex":[0.99837,0.0003246902,0.000238404,0.0003689612,0.0006028358,0.00009520317],"domain_scores_gemma":[0.9980766,0.0009358319,0.000125579,0.0002555058,0.0005572044,0.00004942787],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002244376,0.0002096426,0.0006378219,0.0001845462,0.0001734085,0.00009267116,0.0001317452,0.08316155,0.007132412,0.007712124,0.005073169,0.8952665],"study_design_scores_gemma":[0.00002102676,0.00006087517,0.0002977091,0.00001527876,0.00003554958,0.00009045174,0.00002472356,0.9872831,0.003679794,0.006965579,0.001506925,0.00001902307],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.004295377,0.0001161815,0.9943673,0.00004097225,0.0000279548,0.00006236302,0.00005849352,0.0007273218,0.0003039762],"genre_scores_gemma":[0.0530475,0.00009761429,0.944985,0.00005493931,0.00003953554,0.0002301462,0.0003605373,0.00008529836,0.0010996],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.002512908,"threshold_uncertainty_score":0.01069665,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01209304195142298,"score_gpt":0.2327446953769658,"score_spread":0.2206516534255429,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}