{"id":"W6891789982","doi":"10.48448/1jt2-gg52","title":"Towards Understanding Large-Scale Discourse Structures in Pre-Trained and Fine-Tuned Language Models","year":2022,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Variety (cybernetics); Discourse analysis; Domain of discourse; Information structure; Language understanding; Language model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001635295,0.0006410751,0.0006844722,0.00226489,0.0004236644,0.0002885664,0.001268712,0.0002501273,0.00453912],"category_scores_gemma":[0.0001539092,0.0005815319,0.00008069927,0.002241201,0.001681435,0.0005351469,0.0009506345,0.0007727336,0.00002970627],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001318217,"about_ca_system_score_gemma":0.0009477712,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001442084,"about_ca_topic_score_gemma":0.02412097,"domain_scores_codex":[0.9949808,0.0001589163,0.0004815068,0.001458192,0.001691281,0.001229307],"domain_scores_gemma":[0.9982958,0.00006761519,0.0003746304,0.0009087263,0.00003668485,0.0003165367],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000715295,0.002180994,0.004473425,0.001300848,0.0004840938,0.001288698,0.1231621,0.02753517,0.02916631,0.5067692,0.288912,0.0140119],"study_design_scores_gemma":[0.01183057,0.0007158185,0.001900443,0.0009651533,0.0003952723,0.0002152712,0.1186854,0.6745906,0.0005767931,0.1607755,0.02307037,0.006278817],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.06207388,0.00655379,0.04288655,0.001690104,0.002176484,0.005404084,0.009380956,0.002952289,0.8668818],"genre_scores_gemma":[0.9089986,0.00006866979,0.01291835,0.0002421739,0.0003761514,0.0000839856,0.000589854,0.001261615,0.07546061],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8469247,"threshold_uncertainty_score":0.9996636,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03144478449659047,"score_gpt":0.3239771923018189,"score_spread":0.2925324078052284,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}