{"id":"W2174703538","doi":"10.1007/978-3-642-41230-1_37","title":"Automatically Training Form Classifiers","year":2013,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Web crawler; Domain (mathematical analysis); Machine learning; Training set; Quality (philosophy); Training (meteorology); Data mining; Artificial intelligence; State (computer science); Information retrieval; World Wide Web; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0009641596,0.0005404875,0.0006765772,0.0009741873,0.0002985024,0.001177677,0.004660645,0.000343579,0.0001071098],"category_scores_gemma":[0.0001630537,0.0004633723,0.0002052199,0.0007425151,0.0006983156,0.001101793,0.001344769,0.0008448027,0.0003173552],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002393862,"about_ca_system_score_gemma":0.0007359095,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002236735,"about_ca_topic_score_gemma":0.00003097755,"domain_scores_codex":[0.9957615,0.00002768895,0.0006333095,0.001585791,0.001124769,0.0008670173],"domain_scores_gemma":[0.9969109,0.0004919202,0.0003005472,0.001805913,0.0001926561,0.0002980777],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[6.171574e-7,0.00001092353,0.000008930372,0.00001883726,0.00001954577,0.00004748951,0.001241397,0.001925809,0.00004453547,0.0342146,0.0001426941,0.9623246],"study_design_scores_gemma":[0.0001432521,0.00009221367,0.00006686012,0.0003050644,0.00001430409,0.00005126875,6.513264e-7,0.8522555,0.00009561328,0.1436005,0.002757489,0.0006173503],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00006539309,0.0000981086,0.9790511,0.001235537,0.0009047202,0.0001860163,0.000005076418,0.0003005305,0.01815359],"genre_scores_gemma":[0.0784701,0.00002271773,0.9171357,0.002060901,0.0004517183,0.00001328591,0.0000123504,0.00004149031,0.001791691],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9617073,"threshold_uncertainty_score":0.9998592,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02991305767942924,"score_gpt":0.2467924122671749,"score_spread":0.2168793545877457,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}