{"id":"W4412887855","doi":"10.18653/v1/2025.findings-acl.1147","title":"CausalLink: An Interactive Evaluation Framework for Causal Reasoning","year":2025,"lang":"en","type":"article","venue":"","topic":"Business Process Modeling and Analysis","field":"Business, Management and Accounting","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute","funders":"","keywords":"Computer science; Human–computer interaction; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03561275,0.002673015,0.0012429,0.004302643,0.001282299,0.005495478,0.00446024,0.002257596,0.01629759],"category_scores_gemma":[0.1166965,0.000968654,0.002065327,0.001409277,0.003955995,0.008468579,0.006868519,0.003795923,0.001608423],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003508153,"about_ca_system_score_gemma":0.005023425,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01203143,"about_ca_topic_score_gemma":0.01204052,"domain_scores_codex":[0.9648573,0.02455364,0.00151046,0.002315996,0.005891448,0.0008712357],"domain_scores_gemma":[0.9211627,0.0619366,0.004217937,0.006159319,0.005017697,0.001505704],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001780167,0.0009970645,0.01045019,0.001199762,0.0005881055,0.0005665213,0.002312479,0.3181551,0.007014704,0.3919681,0.01562792,0.2493399],"study_design_scores_gemma":[0.0001260723,0.000260012,0.0004356483,0.0001299846,0.00006899184,0.00008072582,0.000194296,0.8534222,0.004453433,0.1329285,0.007823062,0.00007705623],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.005023947,0.0001235425,0.9843585,0.0005691677,0.00004977223,0.0003777536,0.0004254667,0.005557722,0.003514038],"genre_scores_gemma":[0.2527002,0.0001315355,0.7424011,0.000331734,0.00006143806,0.001056107,0.0007614182,0.001264337,0.001292208],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.03561275,"threshold_uncertainty_score":0.1883405,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03679333588295106,"score_gpt":0.3435157282889484,"score_spread":0.3067223924059973,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}