{"id":"W2009759761","doi":"10.1016/j.ins.2012.07.022","title":"Effectiveness of template detection on noise reduction and websites summarization","year":2012,"lang":"en","type":"article","venue":"Information Sciences","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Automatic summarization; Web page; Information retrieval; The Internet; Noise (video); Preprocessor; Template; Data mining; World Wide Web; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002886427,0.001381806,0.002207182,0.00449686,0.0008996851,0.003074705,0.001328538,0.001603899,0.002830096],"category_scores_gemma":[0.01893155,0.0003639235,0.001641681,0.002840154,0.0004496224,0.002501145,0.0009562287,0.0007587228,0.002987654],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005175191,"about_ca_system_score_gemma":0.002138438,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003748014,"about_ca_topic_score_gemma":0.003081591,"domain_scores_codex":[0.995265,0.001249221,0.0004026261,0.001001267,0.001698165,0.000383648],"domain_scores_gemma":[0.9894729,0.005641414,0.0006752516,0.001465453,0.002332633,0.0004122241],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001586306,0.0004589563,0.006749255,0.0004620608,0.0001904455,0.000183398,0.0001703153,0.0141389,0.05136439,0.001242225,0.006483447,0.9169704],"study_design_scores_gemma":[0.000136508,0.0009141666,0.01782409,0.0000586351,0.0006510583,0.0007824994,0.0004810401,0.8211321,0.1468453,0.003926788,0.007148025,0.00009974917],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3062531,0.004042516,0.6655293,0.0008269849,0.0007202015,0.0004238657,0.002021568,0.01348051,0.006701838],"genre_scores_gemma":[0.6311443,0.001215619,0.3553303,0.0002190813,0.0003789773,0.0001598485,0.005384502,0.0006846861,0.005482738],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00449686,"threshold_uncertainty_score":0.01526511,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0160012018909099,"score_gpt":0.2569754104418357,"score_spread":0.2409742085509258,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}