{"id":"W4283822869","doi":"10.1609/aaai.v36i6.20563","title":"Active Sampling for Text Classification with Subinstance Level Queries","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Machine Learning and Algorithms","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Institute for Catastrophic Loss Reduction; University of Wisconsin-Madison","keywords":"Computer science; Sample (material); Labeled data; Artificial intelligence; Annotation; Machine learning; Salient; Class (philosophy); Domain (mathematical analysis); Selection (genetic algorithm); Task (project management); Data mining; Information retrieval","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007830972,0.001523849,0.002868957,0.001928636,0.00104765,0.003154207,0.004614508,0.002827929,0.003362765],"category_scores_gemma":[0.02426124,0.0007349472,0.00135651,0.00274521,0.001699949,0.005312563,0.00239404,0.003536577,0.001562498],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001830617,"about_ca_system_score_gemma":0.001592513,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002669349,"about_ca_topic_score_gemma":0.003528426,"domain_scores_codex":[0.9944224,0.002586193,0.0003066026,0.00116398,0.001220736,0.0003001879],"domain_scores_gemma":[0.9819887,0.01365467,0.0008345409,0.001793817,0.001258512,0.0004697291],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001375137,0.0009596645,0.004330903,0.0004757457,0.0001436595,0.0002871793,0.0008561186,0.4264496,0.009899519,0.06304444,0.01503159,0.4771464],"study_design_scores_gemma":[0.0000330981,0.00004349834,0.0001101368,0.000007369496,0.000006341704,0.00002270714,0.00002533669,0.9738008,0.0008828014,0.02391364,0.001147292,0.000007070001],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01020814,0.0006008699,0.9858347,0.0004657829,0.00005446611,0.0001865775,0.0002201277,0.001529058,0.0009002748],"genre_scores_gemma":[0.39578,0.0005742026,0.5938556,0.0009600855,0.0007566781,0.001187709,0.002510469,0.0004981298,0.003877141],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007830972,"threshold_uncertainty_score":0.04141468,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1621162943112375,"score_gpt":0.3247130027261861,"score_spread":0.1625967084149486,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}