{"id":"W3090131812","doi":"10.18653/v1/2023.nllp-1.10","title":"Beyond The Text: Analysis of Privacy Statements through Syntactic and Semantic Role Labeling","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Privacy, Security, and Data Protection","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"","keywords":"Computer science; Task (project management); Domain (mathematical analysis); Recall; Natural language processing; Privacy policy; F1 score; Precision and recall; Information privacy; Information retrieval; Artificial intelligence; Internet privacy; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01183308,0.001032329,0.0007616752,0.005270893,0.001838473,0.004378849,0.002270509,0.001947214,0.002873936],"category_scores_gemma":[0.04460983,0.0006351041,0.001497122,0.003118298,0.00360816,0.01676542,0.003673643,0.004061775,0.001859237],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00199062,"about_ca_system_score_gemma":0.003454372,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004696441,"about_ca_topic_score_gemma":0.003289107,"domain_scores_codex":[0.9862737,0.007169971,0.0008490392,0.001996278,0.003253716,0.0004572995],"domain_scores_gemma":[0.9579819,0.02492527,0.005071023,0.007191564,0.004344537,0.0004856396],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","study_design_scores_codex":[0.000600472,0.000463889,0.02043154,0.001198685,0.0001647927,0.001163598,0.01136158,0.02905132,0.02228337,0.4616724,0.01592712,0.4356813],"study_design_scores_gemma":[0.0000481946,0.000117945,0.004057255,0.000374965,0.0001768335,0.0008972388,0.003374257,0.3617986,0.04609539,0.5088783,0.07404391,0.0001371602],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03190591,0.0005949974,0.9518571,0.003861278,0.0001178677,0.0002324743,0.001383151,0.002230897,0.007816236],"genre_scores_gemma":[0.4932584,0.0007690285,0.4963965,0.0009234342,0.0002670341,0.0002381651,0.004166401,0.0007110228,0.00327006],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01183308,"threshold_uncertainty_score":0.06257999,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05657710456648832,"score_gpt":0.3766990392892843,"score_spread":0.320121934722796,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}