{"id":"W3193068792","doi":"10.1162/coli_a_00422","title":"Probing Classifiers: Promises, Shortcomings, and Advances","year":2021,"lang":"en","type":"preprint","venue":"Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Azrieli Foundation; Israel Science Foundation","keywords":"Computer science; Artificial intelligence; Variety (cybernetics); Classifier (UML); Machine learning; Property (philosophy); Artificial neural network; Deep neural networks; Natural language processing; Epistemology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03587117,0.001675193,0.002665594,0.003246215,0.001899018,0.007862362,0.004325789,0.005503535,0.003792456],"category_scores_gemma":[0.09453927,0.001011254,0.001144988,0.003755189,0.004837242,0.02449579,0.003842504,0.01181683,0.003270743],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003302902,"about_ca_system_score_gemma":0.00318854,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004203292,"about_ca_topic_score_gemma":0.001786568,"domain_scores_codex":[0.98483,0.007672116,0.0004602659,0.002329351,0.004201488,0.000506747],"domain_scores_gemma":[0.9011201,0.0717757,0.001939408,0.009236936,0.01419413,0.00173382],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002787897,0.0001345977,0.003670814,0.0005024605,0.0001354047,0.00008146407,0.0006282135,0.0166449,0.0009037099,0.4699956,0.04591766,0.4611063],"study_design_scores_gemma":[0.00002460462,0.00007951927,0.0006542942,0.0003222412,0.00005599769,0.0001294786,0.0002350286,0.1799554,0.001206681,0.7693937,0.04788057,0.00006249973],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0174049,0.09857094,0.7594551,0.09758005,0.00273385,0.0001559836,0.0005860307,0.00252346,0.02098971],"genre_scores_gemma":[0.5606536,0.06184852,0.3392277,0.01191531,0.01477395,0.0005718343,0.001094398,0.00100626,0.008908446],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03587117,"threshold_uncertainty_score":0.1897071,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03569677429684543,"score_gpt":0.2887908698049079,"score_spread":0.2530940955080625,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}