{"id":"W2092683383","doi":"10.1145/584792.584867","title":"Discovering approximate keys in XML data","year":2002,"lang":"en","type":"article","venue":"","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; XML; Key (lock); Information retrieval; Document Structure Description; Set (abstract data type); Representation (politics); XML validation; XML database; Well-formed document; Streaming XML; Search engine indexing; XML Encryption; Data mining; Theoretical computer science; World Wide Web; Programming language; Computer security","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003760541,0.0007052103,0.001438577,0.004134993,0.0009713572,0.003061572,0.001583114,0.001449997,0.0007466087],"category_scores_gemma":[0.03079224,0.0008660012,0.001118623,0.00634867,0.001514473,0.01203719,0.003271945,0.001530817,0.000520682],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009059596,"about_ca_system_score_gemma":0.001143562,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001397575,"about_ca_topic_score_gemma":0.001459474,"domain_scores_codex":[0.9940642,0.001420105,0.001010228,0.0009016186,0.002300971,0.0003030185],"domain_scores_gemma":[0.9809384,0.01188784,0.002525003,0.003073531,0.001374704,0.0002004714],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00101974,0.0001737774,0.02062512,0.00152249,0.0002506489,0.002000621,0.003108935,0.187805,0.02841846,0.1853377,0.007581877,0.5621557],"study_design_scores_gemma":[0.00006684147,0.0001998994,0.002254344,0.0001793986,0.0001212495,0.001975114,0.00108377,0.6136177,0.02668636,0.3329354,0.02078653,0.00009334696],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05564544,0.001444249,0.9400022,0.0006052309,0.00003660272,0.0001381704,0.00104027,0.0006237681,0.0004641033],"genre_scores_gemma":[0.2826438,0.001587858,0.7114523,0.0001982805,0.00009359303,0.0002953286,0.002656014,0.0001212684,0.0009515771],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004134993,"threshold_uncertainty_score":0.01988792,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08292524527261654,"score_gpt":0.2532951234127986,"score_spread":0.1703698781401821,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}