{"id":"W4407830444","doi":"10.1371/journal.pone.0318421","title":"Behind the mask: Random and selective masking in transformer models applied to specialized social science texts","year":2025,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"","keywords":"Masking (illustration); Computer science; Transformer; Classifier (UML); Artificial intelligence; Random forest; Coronavirus disease 2019 (COVID-19); Machine learning; Natural language processing; Speech recognition; Pattern recognition (psychology); Engineering; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007808859,0.001221402,0.0008590344,0.001277603,0.0008010314,0.001788669,0.001388279,0.001376721,0.003481879],"category_scores_gemma":[0.02676248,0.0005347074,0.001273146,0.0006367933,0.001021534,0.003854056,0.001737364,0.00238696,0.002046856],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000946199,"about_ca_system_score_gemma":0.001118855,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003213045,"about_ca_topic_score_gemma":0.005318031,"domain_scores_codex":[0.9979971,0.001182513,0.0001204942,0.0003352952,0.0002429717,0.000121613],"domain_scores_gemma":[0.9880744,0.009253356,0.0006132726,0.001259092,0.0005759048,0.0002240535],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005312745,0.0008551054,0.04387574,0.0007812227,0.0006115512,0.0006243908,0.00266076,0.2993419,0.02675841,0.04623896,0.01591321,0.5570259],"study_design_scores_gemma":[0.00007300371,0.0003023401,0.002209858,0.00005446315,0.00007243255,0.0001222605,0.0001644301,0.9616515,0.006985702,0.02642586,0.001898519,0.00003964387],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5012184,0.001360856,0.4814941,0.002465587,0.0002794336,0.0005467945,0.001120961,0.004790903,0.006722915],"genre_scores_gemma":[0.9064509,0.0002264134,0.08864444,0.0004437066,0.0000957516,0.0001958862,0.001144383,0.0002289283,0.002569491],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.007808859,"threshold_uncertainty_score":0.04129773,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05780898429034564,"score_gpt":0.3366705908533048,"score_spread":0.2788616065629591,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}