{"id":"W7130543099","doi":"10.1109/fllm67465.2025.11391103","title":"Agent-guided Causal Discovery with a Small Language Model","year":2025,"lang":"","type":"article","venue":"","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Spurious relationship; Causal model; Language model; Equivalence (formal languages); Task (project management); Causal structure; Causal inference; Benchmark (surveying)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0005859472,0.0005689911,0.0005196348,0.0003989087,0.0004482595,0.001702709,0.002113354,0.000198678,0.0001361841],"category_scores_gemma":[0.0001311699,0.0004781488,0.0001838439,0.001779359,0.0003088341,0.001877641,0.001173055,0.0004378816,0.0003714849],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002785265,"about_ca_system_score_gemma":0.00137753,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00264817,"about_ca_topic_score_gemma":0.003516628,"domain_scores_codex":[0.9958681,0.0001599902,0.000797003,0.001408866,0.0005450724,0.001220923],"domain_scores_gemma":[0.9971086,0.0001920441,0.00018116,0.001959461,0.0003038008,0.0002548716],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001327795,0.0006316205,0.0005148069,0.0001962791,0.0002376852,0.0006165481,0.01062217,0.1112837,0.006419592,0.8238841,0.006945826,0.03851495],"study_design_scores_gemma":[0.000299649,0.0001933778,0.00006154877,0.0002391952,0.00006950964,0.00001929479,0.002009013,0.9109321,0.07801414,0.006868807,0.0006767093,0.0006166799],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07999989,0.0004529926,0.8478968,0.003086803,0.0006231258,0.0006126085,0.000009034486,0.0002403112,0.06707851],"genre_scores_gemma":[0.7832212,0.00006132825,0.04928572,0.003116,0.00008199771,0.00005151762,0.000004029134,0.00003058328,0.1641476],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8170152,"threshold_uncertainty_score":0.999767,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04936350075707562,"score_gpt":0.2976340889346764,"score_spread":0.2482705881776008,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}