{"id":"W1986114210","doi":"10.1016/j.csda.2008.08.001","title":"Finding approximate solutions to combinatorial problems with very large data sets using BIRCH","year":2008,"lang":"en","type":"article","venue":"Computational Statistics & Data Analysis","topic":"Advanced Statistical Methods and Models","field":"Mathematics","cited_by":11,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of British Columbia","funders":"TRIUMF","keywords":"Outlier; Estimator; Computer science; Algorithm; Robustness (evolution); Covariance; Mathematical optimization; Mathematics; Statistics; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008062643,0.001725891,0.003223975,0.005930732,0.001836676,0.002828894,0.002766329,0.002171292,0.005361264],"category_scores_gemma":[0.03961556,0.001728562,0.002884257,0.006093862,0.003813771,0.006102058,0.004933638,0.004926791,0.00126438],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001822481,"about_ca_system_score_gemma":0.002688305,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005245429,"about_ca_topic_score_gemma":0.005519119,"domain_scores_codex":[0.9951255,0.002564457,0.000276474,0.0006591301,0.001064903,0.0003094946],"domain_scores_gemma":[0.9805994,0.0150973,0.0006556934,0.002421276,0.0009061133,0.000320257],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000358896,0.0003244847,0.002678794,0.0007002018,0.0004023188,0.0002162313,0.000381737,0.3426213,0.002059984,0.4150126,0.01186116,0.2233823],"study_design_scores_gemma":[0.00004145158,0.00006087394,0.0002789058,0.0000412854,0.00004007075,0.00007325488,0.00004793102,0.6188744,0.0005221431,0.3781565,0.001839238,0.00002389298],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.005272169,0.000652965,0.9920176,0.0003076289,0.00007497372,0.00007466875,0.0000782686,0.0003554971,0.001166209],"genre_scores_gemma":[0.1446947,0.001310787,0.8480535,0.000407198,0.0002583267,0.000654057,0.0008124306,0.000339056,0.003470074],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008062643,"threshold_uncertainty_score":0.04263985,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3862878097668099,"score_gpt":0.463281868573303,"score_spread":0.07699405880649307,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}