{"id":"W4404390328","doi":"10.48550/arxiv.2411.07180","title":"Gumbel Counterfactual Generation From Language Models","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Institute for Catastrophic Loss Reduction","keywords":"Counterfactual thinking; Computer science; Linguistics; Natural language processing; Psychology; Philosophy; Social psychology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01495408,0.001405151,0.00134202,0.001584018,0.001026224,0.003092938,0.00241149,0.001785713,0.007480238],"category_scores_gemma":[0.06233414,0.00110635,0.00258108,0.0012596,0.003529761,0.005055586,0.003634741,0.0045226,0.0009056312],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002539748,"about_ca_system_score_gemma":0.001904897,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002343781,"about_ca_topic_score_gemma":0.003203639,"domain_scores_codex":[0.9900694,0.006613195,0.000323468,0.001290474,0.001368186,0.000335177],"domain_scores_gemma":[0.9466544,0.04667011,0.001655127,0.003652661,0.001031454,0.0003361426],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003667713,0.0001413132,0.001678701,0.0002068285,0.0001375537,0.0002470667,0.000701318,0.3585819,0.001634114,0.5656229,0.002478245,0.06820323],"study_design_scores_gemma":[0.00004261215,0.00002436696,0.00008056952,0.0000197637,0.00001480126,0.00003315038,0.00002183542,0.7341861,0.0009066989,0.2637846,0.0008688186,0.00001673588],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01256686,0.0001121857,0.984674,0.0004395985,0.00003017163,0.0001119456,0.0001187895,0.000491099,0.001455328],"genre_scores_gemma":[0.4128053,0.000248083,0.5811378,0.0005048902,0.00008781473,0.0008716122,0.0005699416,0.0003908657,0.003383615],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01495408,"threshold_uncertainty_score":0.07908565,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07081071935022155,"score_gpt":0.2103113184266946,"score_spread":0.1395005990764731,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}