{"id":"W4412887855","doi":"10.18653/v1/2025.findings-acl.1147","title":"CausalLink: An Interactive Evaluation Framework for Causal Reasoning","year":2025,"lang":"en","type":"article","venue":"","topic":"Business Process Modeling and Analysis","field":"Business, Management and Accounting","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute","funders":"","keywords":"Computer science; Human–computer interaction; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006846912,0.0001504065,0.0001913841,0.0003232031,0.0002603852,0.0004266056,0.0001799819,0.0001016251,0.0002090142],"category_scores_gemma":[0.0007831327,0.0001342299,0.00007942762,0.0006688244,0.0000193589,0.001328754,0.00006287517,0.0001273673,0.00003209456],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004218795,"about_ca_system_score_gemma":0.00005926419,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000575651,"about_ca_topic_score_gemma":0.0001842823,"domain_scores_codex":[0.9989718,0.000009868102,0.0002347347,0.0003600469,0.000213387,0.000210122],"domain_scores_gemma":[0.9985605,0.00008849325,0.000138741,0.000225601,0.0009786512,0.000007961229],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001824331,0.0002373674,0.008050269,0.0003262531,0.0002424245,0.000001055577,0.0001236991,0.01332384,0.000161605,0.8290148,0.002515465,0.1458208],"study_design_scores_gemma":[0.0003565025,0.000004537478,0.0009190925,0.0001776244,0.0003752593,1.161643e-7,0.000499642,0.7995002,0.0000569061,0.1945055,0.003424225,0.0001803673],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06069806,0.00009709392,0.9239718,0.001834128,0.0004949722,0.0003143465,0.000001265232,0.0002386012,0.01234979],"genre_scores_gemma":[0.9813058,0.00000367901,0.01460485,0.002439879,0.0008117938,0.0001258761,0.0001099003,0.00001901486,0.0005792095],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9206077,"threshold_uncertainty_score":0.5473738,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03679333588295106,"score_gpt":0.3435157282889484,"score_spread":0.3067223924059973,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}