{"id":"W2296700935","doi":"","title":"Grammar Inference for Web Documents","year":2011,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"XML validation; Computer science; Presentational and representational acting; Document Structure Description; XML Schema Editor; Efficient XML Interchange; Streaming XML; XML Schema (W3C); Information retrieval; SGML; XML Encryption; XML; XML Signature; Programming language; Natural language processing; World Wide Web; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006397826,0.001107456,0.001292928,0.004836239,0.00163597,0.003306096,0.003264335,0.002326147,0.005930692],"category_scores_gemma":[0.04129652,0.001284449,0.004018066,0.003236752,0.002403297,0.005620159,0.002665783,0.003329031,0.002562406],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002259806,"about_ca_system_score_gemma":0.002571851,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008591332,"about_ca_topic_score_gemma":0.008566297,"domain_scores_codex":[0.9924647,0.002702436,0.0007152692,0.00174202,0.002108417,0.0002671081],"domain_scores_gemma":[0.9722075,0.0224317,0.000948169,0.002609222,0.001551496,0.0002518918],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002153018,0.0002022064,0.005501926,0.0008446401,0.0003561822,0.0009475566,0.0009874109,0.2873624,0.003659865,0.2818397,0.01708527,0.4009975],"study_design_scores_gemma":[0.00004207528,0.00001896391,0.0003693622,0.00008033343,0.0000405299,0.0002880857,0.00007789318,0.611073,0.002440892,0.3759027,0.009633627,0.00003247644],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.003180645,0.0001883573,0.9919093,0.0002605449,0.00003246245,0.0001042987,0.0006065518,0.002575562,0.001142233],"genre_scores_gemma":[0.1052758,0.0005265048,0.884097,0.000422981,0.0001266835,0.0003614096,0.005553516,0.000932241,0.002703755],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008591332,"threshold_uncertainty_score":0.03383535,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04948157682302046,"score_gpt":0.2802402359717129,"score_spread":0.2307586591486924,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}