{"id":"W4393622469","doi":"10.5281/zenodo.7812052","title":"Data from: The problem with missingness in mixed-type datasets and what we can do about it","year":2023,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Advanced Clustering Algorithms Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Guelph","funders":"","keywords":"Missing data; Type (biology); Statistics; Computer science; Mathematics; Data science; Biology; Ecology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01787831,0.001664313,0.002172961,0.002874048,0.001955243,0.005757481,0.004909798,0.005063353,0.03793137],"category_scores_gemma":[0.1090841,0.001517432,0.002759578,0.006823377,0.001833453,0.004653007,0.003596932,0.003983813,0.02786556],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001575689,"about_ca_system_score_gemma":0.00400961,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004285615,"about_ca_topic_score_gemma":0.009658647,"domain_scores_codex":[0.9886535,0.003474888,0.002230355,0.00223358,0.002809994,0.0005977658],"domain_scores_gemma":[0.938447,0.0330392,0.003298082,0.01856368,0.005619434,0.001032627],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002429014,0.00004831001,0.002685243,0.002258216,0.0001540456,0.00007623511,0.00009438752,0.0008218855,0.0003924453,0.002675385,0.9820688,0.008482116],"study_design_scores_gemma":[0.001171727,0.00008058071,0.00815767,0.001501188,0.0001558326,0.0005195068,0.0003908119,0.004743556,0.002892963,0.03082501,0.9493752,0.0001859649],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.001788022,0.0008790987,0.005557866,0.00362009,0.0007398931,0.0001693986,0.9826187,0.003256051,0.001370908],"genre_scores_gemma":[0.006893065,0.0003419392,0.01819395,0.00146709,0.0001635704,0.0009644704,0.9693171,0.001220703,0.001438151],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.9821217,"threshold_uncertainty_score":0.1268931,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0762632156264673,"score_gpt":0.306553596723691,"score_spread":0.2302903810972237,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}