{"id":"W7106029918","doi":"10.7939/83025","title":"Big Data Analytics: Methodology Evaluation and Development for Gene Set Analysis &amp; Spatial Cluster Detection in Administrative Health Data","year":2025,"lang":"en","type":"dissertation","venue":"University of Alberta Library","topic":"Gene expression and cancer classification","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Big data; Covariance; Set (abstract data type); Lasso (programming language); Data set; Covariance matrix; Stability (learning theory); Estimation of covariance matrices; Type I and type II errors","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03693038,0.001353642,0.001104606,0.005417933,0.001096167,0.003976475,0.002766118,0.001281335,0.003711419],"category_scores_gemma":[0.06435771,0.0008271225,0.001762019,0.005911529,0.0009481239,0.003455615,0.003470989,0.002526731,0.001693004],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001798692,"about_ca_system_score_gemma":0.005433487,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005960345,"about_ca_topic_score_gemma":0.004653645,"domain_scores_codex":[0.9804255,0.01143984,0.001184147,0.001332842,0.005229658,0.0003879768],"domain_scores_gemma":[0.9525826,0.02480352,0.00212804,0.00658129,0.01280427,0.00110021],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008007809,0.0006749585,0.02241679,0.001365918,0.0007262601,0.0002787623,0.0006475805,0.07664224,0.006446177,0.04960009,0.05986962,0.7805308],"study_design_scores_gemma":[0.0001503002,0.0003035227,0.005405022,0.000311729,0.00009693623,0.0002011487,0.0004940653,0.9237503,0.009781734,0.02714443,0.03228343,0.0000774837],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01436304,0.00212917,0.9653087,0.002619854,0.0004312068,0.001143035,0.002459139,0.008055646,0.003490145],"genre_scores_gemma":[0.06498075,0.0009981332,0.9284225,0.0005052322,0.0001227603,0.001252675,0.002588456,0.0005350194,0.0005944368],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.03693038,"threshold_uncertainty_score":0.1953089,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2113712566999396,"score_gpt":0.3720180530734725,"score_spread":0.1606467963735329,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}