{"id":"W4415282304","doi":"10.3390/info16100904","title":"Can We Trust AI Content Detection Tools for Critical Decision-Making?","year":2025,"lang":"en","type":"article","venue":"Information","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia, Okanagan Campus; Government of Manitoba","funders":"","keywords":"Government (linguistics); Robustness (evolution); Precision and recall; Recall; Reliability (semiconductor); Content analysis","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1262562,0.002092443,0.002142053,0.01079809,0.004293248,0.0334196,0.00563793,0.007781909,0.01226042],"category_scores_gemma":[0.5031828,0.001794766,0.001306695,0.004792881,0.01479751,0.05855674,0.01206974,0.007894453,0.01352893],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006201313,"about_ca_system_score_gemma":0.01491232,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005298702,"about_ca_topic_score_gemma":0.003547417,"domain_scores_codex":[0.8939543,0.06216814,0.007032021,0.01129636,0.02186631,0.003682897],"domain_scores_gemma":[0.3936056,0.4071891,0.03560981,0.08887563,0.06192673,0.01279304],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001201716,0.0004250114,0.02881881,0.002916781,0.0004033667,0.0006323291,0.02308123,0.005741048,0.00487865,0.1735838,0.09391791,0.6643994],"study_design_scores_gemma":[0.0002711696,0.0004201001,0.008054772,0.003529258,0.0002432438,0.0009307205,0.01112924,0.04021442,0.01009017,0.6074463,0.3171745,0.0004961296],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06283611,0.01634392,0.5458488,0.2536694,0.004540834,0.001041783,0.001717216,0.01963435,0.09436768],"genre_scores_gemma":[0.5694838,0.003963023,0.3940021,0.01831372,0.001720944,0.0009057235,0.00131628,0.002969218,0.007325185],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1262562,"threshold_uncertainty_score":0.6677147,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03774558943092063,"score_gpt":0.3273950318151122,"score_spread":0.2896494423841915,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}