{"id":"W2533861448","doi":"10.1145/2983323.2983776","title":"Scalability of Continuous Active Learning for Reliable High-Recall Text Classification","year":2016,"lang":"en","type":"article","venue":"","topic":"Machine Learning and Algorithms","field":"Computer Science","cited_by":80,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Scalability; Classifier (UML); Limiting; Computer science; Recall; Artificial intelligence; Machine learning; Binary logarithm; Precision and recall; Class (philosophy); Information retrieval; Mathematics; Discrete mathematics; Database","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01156597,0.001625118,0.001917777,0.002086417,0.0009469083,0.003515176,0.005207991,0.002492375,0.003839417],"category_scores_gemma":[0.04534916,0.001023922,0.001016799,0.002024318,0.002286736,0.007220058,0.004738386,0.003560989,0.002283967],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001729518,"about_ca_system_score_gemma":0.001827554,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003342862,"about_ca_topic_score_gemma":0.003131978,"domain_scores_codex":[0.9933831,0.001970976,0.0003916211,0.001438715,0.002501895,0.0003136603],"domain_scores_gemma":[0.9689741,0.02023782,0.001288891,0.005603043,0.003244341,0.0006518493],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001189237,0.0006634304,0.004407856,0.0005225536,0.0001713523,0.0002780333,0.0005261101,0.3180613,0.01363162,0.01969392,0.0119768,0.6288778],"study_design_scores_gemma":[0.00005066864,0.00005209866,0.0002780664,0.00001322762,0.00001070722,0.00005443837,0.00003355108,0.9835159,0.002423139,0.01261725,0.0009388205,0.00001210084],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03814657,0.001385567,0.9471282,0.001111782,0.0001180941,0.0002658803,0.0003891965,0.007909627,0.003545131],"genre_scores_gemma":[0.6040124,0.0005846552,0.3896445,0.000535611,0.0003986055,0.0006506123,0.001389188,0.0004656987,0.002318758],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01156597,"threshold_uncertainty_score":0.06116742,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0129889152481545,"score_gpt":0.2532292927712201,"score_spread":0.2402403775230657,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}