{"id":"W2979333226","doi":"10.1109/ccece.2019.8861966","title":"Semiparametric Subsampling and Data Condensation for Large-Scale Data Analytics","year":2019,"lang":"en","type":"article","venue":"","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University; University of Waterloo","funders":"","keywords":"Computer science; Cluster analysis; Data mining; Benchmark (surveying); Artificial intelligence; Scalability; Pattern recognition (psychology); Random forest; Voronoi diagram; Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01011977,0.001277649,0.001619324,0.002128938,0.0008220968,0.001585268,0.002311039,0.001165731,0.001269371],"category_scores_gemma":[0.02587344,0.000687417,0.001954276,0.002038205,0.002020989,0.002255368,0.002850541,0.002073009,0.0006031426],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001007853,"about_ca_system_score_gemma":0.001792184,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002796921,"about_ca_topic_score_gemma":0.004175319,"domain_scores_codex":[0.99501,0.002583643,0.0003027828,0.0009237684,0.001025177,0.0001546068],"domain_scores_gemma":[0.9883196,0.006564956,0.0008129574,0.003023071,0.001061634,0.0002177988],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006869074,0.0003786549,0.01462009,0.0007666745,0.0008307341,0.0004506705,0.001305564,0.3561487,0.04333797,0.1056981,0.007630802,0.4681451],"study_design_scores_gemma":[0.00003678566,0.0001447658,0.002630057,0.00003989931,0.00006447832,0.0001617055,0.000113884,0.9195473,0.01318552,0.05847029,0.005544693,0.00006062767],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0040724,0.0001769999,0.994893,0.00008207838,0.00002206263,0.00008886918,0.0001118117,0.0004120565,0.000140783],"genre_scores_gemma":[0.1612498,0.0003449488,0.8352249,0.000207688,0.0001620449,0.000655462,0.001378743,0.0002655726,0.0005108919],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01011977,"threshold_uncertainty_score":0.05351907,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1126477890387248,"score_gpt":0.3516509776899865,"score_spread":0.2390031886512617,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}