{"id":"W2167435152","doi":"10.1145/1242572.1242838","title":"Adaptive record extraction from web pages","year":2007,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Cluster analysis; Matching (statistics); Tree (set theory); Metric (unit); Web page; Data mining; Information retrieval; World Wide Web; Artificial intelligence; Engineering; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001753573,0.0008071263,0.001306489,0.008197604,0.0008972299,0.002430798,0.002686631,0.001330561,0.002601343],"category_scores_gemma":[0.01336951,0.0006576242,0.001289111,0.01292684,0.0005850797,0.003325647,0.002084091,0.00139293,0.003790177],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003772423,"about_ca_system_score_gemma":0.001592541,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001941251,"about_ca_topic_score_gemma":0.004124977,"domain_scores_codex":[0.9959462,0.0004860813,0.000616975,0.0009246269,0.001857872,0.0001682021],"domain_scores_gemma":[0.9892159,0.003393144,0.00116091,0.003477554,0.002539562,0.0002128337],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0001604604,0.0002874292,0.009413445,0.000743253,0.00018035,0.0002950681,0.0003926848,0.004016516,0.04603907,0.003500025,0.01181869,0.9231529],"study_design_scores_gemma":[0.0001599132,0.0006000651,0.03427827,0.0003222591,0.0007374696,0.005336263,0.0009469673,0.4537959,0.2521161,0.04151258,0.2097587,0.0004355923],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01642769,0.000614106,0.9709947,0.000199065,0.0000847326,0.0004251066,0.002860898,0.007187145,0.001206631],"genre_scores_gemma":[0.03587163,0.0003742429,0.9538886,0.000108178,0.0000739724,0.0002987112,0.007009118,0.0003855171,0.001990026],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008197604,"threshold_uncertainty_score":0.009273887,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03173912972544339,"score_gpt":0.2678346333360881,"score_spread":0.2360955036106447,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}