{"id":"W107766491","doi":"10.1007/978-3-319-09912-5_1","title":"A Semi-supervised Learning Algorithm for Web Information Extraction with Tolerance Rough Sets","year":2014,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Rough Sets and Fuzzy Logic","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Winnipeg","funders":"","keywords":"Computer science; Categorical variable; Noun phrase; Artificial intelligence; Phrase; Natural language processing; Vector space model; Document classification; Machine learning; Information extraction; Naive Bayes classifier; Representation (politics); Data mining; Algorithm; Noun; Support vector machine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.000829699,0.0005170243,0.0005008419,0.0005292943,0.0004461788,0.0009388402,0.001740645,0.0003269229,0.000007813377],"category_scores_gemma":[0.0000471587,0.0004198237,0.0001165538,0.0004883579,0.0002628369,0.001803124,0.0003671432,0.0007761715,0.00003449525],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002176668,"about_ca_system_score_gemma":0.0004064082,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001460751,"about_ca_topic_score_gemma":0.00001968006,"domain_scores_codex":[0.9969164,0.00003570936,0.0005185528,0.00101321,0.0008673561,0.0006487562],"domain_scores_gemma":[0.9978197,0.0003904001,0.000431917,0.0008358931,0.0003790032,0.000143137],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000008243921,0.000008646468,0.00001197922,0.00003789701,0.000006291562,0.000007441814,0.0004767339,0.06870843,0.00001110267,0.001201114,0.00003342155,0.9294887],"study_design_scores_gemma":[0.000569544,0.0004436408,0.00005679229,0.0002703859,0.000008026975,0.00009269253,3.390827e-7,0.9699926,0.0000800167,0.008862384,0.01905005,0.000573552],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00004408348,0.0001200187,0.9947421,0.0004110049,0.001211864,0.0006743862,0.000008835052,0.0002421255,0.002545564],"genre_scores_gemma":[0.03021433,0.00003572666,0.9675034,0.001570874,0.0004694737,0.00004246482,0.00003445959,0.00002852485,0.0001007633],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9289151,"threshold_uncertainty_score":0.9998254,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01209304195142298,"score_gpt":0.2327446953769658,"score_spread":0.2206516534255429,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}