{"id":"W3048081621","doi":"10.48550/arxiv.2008.02954","title":"Deep Active Learning with Crowdsourcing Data for Privacy Policy Classification","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Mobile Crowdsensing and Crowdsourcing","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Crowdsourcing; Computer science; Leverage (statistics); Classifier (UML); Machine learning; Artificial intelligence; Privacy policy; Annotation; Labeled data; Class (philosophy); Active learning (machine learning); Information privacy; World Wide Web; Internet privacy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0003188476,0.0004172983,0.0004337433,0.0003087169,0.0005113953,0.0004182023,0.003412536,0.0002718716,0.00000345532],"category_scores_gemma":[0.0003556334,0.0004573199,0.0001357774,0.0008546877,0.0001236298,0.0007212211,0.003764646,0.0009465754,0.000023137],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003282315,"about_ca_system_score_gemma":0.00063354,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001903615,"about_ca_topic_score_gemma":0.00003210844,"domain_scores_codex":[0.996797,0.0001856674,0.0002395836,0.002124453,0.0001415439,0.0005117042],"domain_scores_gemma":[0.9960043,0.0002828992,0.0005200245,0.002683298,0.0002545781,0.0002549037],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003282733,0.0001448078,0.00213455,0.0004716503,0.0004579496,0.0002187543,0.004175178,0.8087709,0.0008983544,0.1568323,0.0004909444,0.02507634],"study_design_scores_gemma":[0.0006050805,0.0001036709,0.001079725,0.0001655373,0.0001075161,0.000008258874,0.0004834865,0.9874304,0.0002028966,0.00597015,0.003291613,0.0005516752],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05259062,0.00002481991,0.944087,0.0009814355,0.0001693311,0.0005481764,0.00002042774,0.0005241702,0.001053995],"genre_scores_gemma":[0.9841733,0.00003510573,0.01468228,0.0001333936,0.0003168712,0.000003143138,0.0001942121,0.00004994204,0.0004117704],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9315827,"threshold_uncertainty_score":0.9997879,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1448672035770451,"score_gpt":0.2307093802426244,"score_spread":0.08584217666557933,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}