{"id":"W4320710261","doi":"10.1080/08989621.2023.2179919","title":"Can ChatGPT be trusted to provide reliable estimates?","year":2023,"lang":"en","type":"letter","venue":"Accountability in Research","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"Thompson Rivers University","funders":"","keywords":"Direct Anonymous Attestation; Computer security; Internet privacy; Computer science; Scientific misconduct; Trusted Computing; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":["metaresearch"],"domain":"methods","study_design":"not_applicable","genre":"commentary","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"gpt","categories":[],"domain":null,"study_design":"not_applicable","genre":"commentary","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"}],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02915644,0.0007440938,0.001103639,0.000957013,0.007514229,0.009106311,0.003119325,0.05955176,0.02090753],"category_scores_gemma":[0.2476809,0.0008691972,0.0009289294,0.0009444882,0.006720143,0.01618481,0.004636007,0.07787675,0.0231584],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005532331,"about_ca_system_score_gemma":0.01163592,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008015739,"about_ca_topic_score_gemma":0.01405033,"domain_scores_codex":[0.9785948,0.006590042,0.001889431,0.00217747,0.00890747,0.00184072],"domain_scores_gemma":[0.7957461,0.1437206,0.009435314,0.007572729,0.02804636,0.01547883],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00003282975,0.00001294378,0.00026394,0.00003501447,0.000004865185,0.0001950908,0.0002638476,0.00002967522,0.0000621224,0.01273951,0.9734958,0.01286431],"study_design_scores_gemma":[0.00003780995,0.00002917283,0.000372834,0.000207275,0.00001008773,0.0002681108,0.0004564147,0.0004398705,0.0002061948,0.01949592,0.9784083,0.00006786406],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.00008023707,0.0004803449,0.0003860569,0.9887267,0.007869287,0.000005601656,0.00002261507,0.00005868002,0.002370499],"genre_scores_gemma":[0.003264136,0.0005940268,0.0008930986,0.9733264,0.01196355,0.00006124056,0.00003066366,0.0001082846,0.009758591],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.9708436,"threshold_uncertainty_score":0.1541958,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2534849222970801,"score_gpt":0.4491510783357754,"score_spread":0.1956661560386953,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}