{"id":"W2068704925","doi":"10.1109/icmla.2013.187","title":"Selective Sampling Designs to Improve the Performance of Classification Methods","year":2013,"lang":"en","type":"article","venue":"","topic":"Bayesian Modeling and Causal Inference","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Naive Bayes classifier; Sampling (signal processing); Logistic regression; Computer science; Statistics; Bayes' theorem; Data mining; Missing data; Binary classification; Artificial intelligence; Bayes error rate; Sample size determination; Machine learning; Mathematics; Bayesian probability; Bayes classifier; Support vector machine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1148802,0.001760618,0.00239928,0.002200807,0.001674641,0.002151022,0.002895046,0.003207044,0.001986662],"category_scores_gemma":[0.270286,0.001298241,0.001757796,0.001695098,0.002373927,0.003919665,0.003256182,0.002962175,0.0007074323],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00150822,"about_ca_system_score_gemma":0.003314957,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00251046,"about_ca_topic_score_gemma":0.002277933,"domain_scores_codex":[0.9166529,0.06863457,0.003626195,0.004565735,0.005732892,0.0007875544],"domain_scores_gemma":[0.69318,0.2545658,0.0099898,0.02519882,0.01592879,0.001136879],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005879088,0.001553701,0.05511961,0.001587194,0.002057222,0.0004567266,0.003718096,0.3305492,0.01670087,0.06963468,0.005912396,0.5068313],"study_design_scores_gemma":[0.00114112,0.001974788,0.005364444,0.0002601226,0.0004830782,0.0002115906,0.0002934231,0.9184738,0.01093191,0.05644992,0.004326857,0.00008895328],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.04800991,0.0006830245,0.9476353,0.0005336648,0.0001363086,0.001412288,0.0001308497,0.0005659233,0.0008926952],"genre_scores_gemma":[0.349467,0.0003907098,0.6450813,0.0006747598,0.0002634485,0.00301776,0.0004189783,0.0001201287,0.0005659576],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1148802,"threshold_uncertainty_score":0.6075519,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.112058794588624,"score_gpt":0.3578968239430307,"score_spread":0.2458380293544068,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}