{"id":"W4226145103","doi":"10.18653/v1/2022.findings-acl.326","title":"On the data requirements of probing","year":2022,"lang":"en","type":"article","venue":"Findings of the Association for Computational Linguistics: ACL 2022","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; University of Toronto","funders":"","keywords":"Computer science; Reliability (semiconductor); Context (archaeology); Construct (python library); Data mining; Machine learning; Artificial intelligence; Power (physics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001935687,0.00008009798,0.0001332732,0.00007202852,0.0004806764,0.00003929638,0.002341199,0.00002351638,0.00001996868],"category_scores_gemma":[0.006392225,0.00006375978,0.00008454422,0.0003089114,0.00001746026,0.00005148807,0.001513183,0.0001723826,0.000001944457],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002726657,"about_ca_system_score_gemma":0.0001566893,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001037034,"about_ca_topic_score_gemma":9.101424e-7,"domain_scores_codex":[0.9981132,0.0001155702,0.0003984049,0.0002598465,0.0009631282,0.0001498619],"domain_scores_gemma":[0.9966307,0.001612452,0.0007406149,0.0005787236,0.0004219993,0.00001552927],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000008756242,0.00006504196,0.001995301,0.00001674788,0.00006538478,1.076651e-7,0.0004120751,0.1644338,0.00004575651,0.8228952,0.009920309,0.0001415276],"study_design_scores_gemma":[0.0003966726,0.00007553019,0.001780904,0.00002582805,0.00002701945,4.873829e-7,0.00005644387,0.7376456,0.0002308488,0.2493767,0.01027458,0.0001094611],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.170128,0.0001543205,0.7727475,0.01828804,0.0168405,0.004351182,0.004406187,0.0002535122,0.01283081],"genre_scores_gemma":[0.9873166,3.824734e-7,0.01160816,0.0003236197,0.0001099008,0.00002616911,0.00009821661,0.000009006897,0.0005079626],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8171886,"threshold_uncertainty_score":0.7652552,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05416273860118753,"score_gpt":0.291028757242492,"score_spread":0.2368660186413045,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}