{"id":"W3200130628","doi":"10.18653/v1/2021.emnlp-main.122","title":"Conditional probing: measuring usable information beyond a baseline","year":2021,"lang":"en","type":"article","venue":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing","topic":"Topic Modeling","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; National Science Foundation","keywords":"Baseline (sea); USable; Representation (politics); Computer science; Word (group theory); Property (philosophy); Identity (music); Natural language processing; Artificial intelligence; Speech recognition; Mathematics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008933778,0.001324921,0.001264233,0.001337442,0.0006482222,0.002312578,0.001650814,0.002263468,0.002194627],"category_scores_gemma":[0.06834812,0.0007824252,0.0009894497,0.001445725,0.00306382,0.01035901,0.003767538,0.004318867,0.0003600494],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001101396,"about_ca_system_score_gemma":0.0007890836,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001113722,"about_ca_topic_score_gemma":0.000939293,"domain_scores_codex":[0.9949567,0.002229617,0.000307234,0.001368005,0.0008264302,0.0003120841],"domain_scores_gemma":[0.9248345,0.05753485,0.004006672,0.01053541,0.001919722,0.001168801],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.006123666,0.001388968,0.1078488,0.001821885,0.00166009,0.0006272247,0.002814818,0.3028664,0.1079394,0.1411985,0.00522641,0.3204839],"study_design_scores_gemma":[0.00009316794,0.001393226,0.02475498,0.0001407305,0.0003884749,0.0003913874,0.0003171033,0.6270015,0.03665104,0.306761,0.001921241,0.0001860263],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3439771,0.0006867602,0.6485376,0.0007606136,0.00005536628,0.0001451458,0.00100681,0.001461698,0.00336884],"genre_scores_gemma":[0.966808,0.0001368951,0.03143514,0.0001994227,0.00003521072,0.0001245323,0.0007592593,0.0001696582,0.000331786],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008933778,"threshold_uncertainty_score":0.04724693,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05992423051901477,"score_gpt":0.3690637550504681,"score_spread":0.3091395245314533,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}