{"id":"W2913760405","doi":"10.48550/arxiv.1901.11040","title":"The Wilderness Area Data Set: Adapting the Covertype data set for unsupervised learning","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Data Stream Mining Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Set (abstract data type); Data set; Cluster analysis; Data stream mining; Context (archaeology); Unsupervised learning; USable; Benchmark (surveying); Data mining; Data stream; Machine learning; Process (computing); Artificial intelligence; Geography; World Wide Web; Cartography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004223083,0.0007247014,0.0006119486,0.003504211,0.001075893,0.001878288,0.002121512,0.001440592,0.001353213],"category_scores_gemma":[0.02420474,0.0002263107,0.001083174,0.005104032,0.001154584,0.002413538,0.0020116,0.00189186,0.0007079375],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009719775,"about_ca_system_score_gemma":0.001254368,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006765479,"about_ca_topic_score_gemma":0.009950751,"domain_scores_codex":[0.9965029,0.0008180814,0.0003531474,0.0005641847,0.001561289,0.0002003824],"domain_scores_gemma":[0.9821485,0.00534676,0.001120735,0.006673566,0.004045995,0.0006644978],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002032403,0.002925,0.208251,0.001050613,0.000584002,0.001312886,0.001391508,0.1798293,0.01702024,0.04252639,0.1874293,0.3556473],"study_design_scores_gemma":[0.0004411773,0.001355913,0.1282636,0.0002118314,0.000128272,0.0019295,0.001620834,0.6341746,0.03406501,0.06093463,0.136602,0.0002725825],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"dataset","genre_scores_codex":[0.7291192,0.0008428704,0.1626109,0.002498972,0.0008670224,0.001365794,0.08085394,0.004894859,0.0169465],"genre_scores_gemma":[0.635833,0.0004625233,0.2207977,0.0005569607,0.0002549993,0.001465083,0.1375303,0.0006270901,0.002472427],"genre_candidate":"dataset","genre_consensus":null,"teacher_disagreement_score":0.006765479,"threshold_uncertainty_score":0.02233404,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3154898749161229,"score_gpt":0.2551777202509075,"score_spread":0.06031215466521539,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}