{"id":"W6911970138","doi":"10.5281/zenodo.15013450","title":"Measurement scales and metrics for trusworthy AI","year":2025,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Ethics and Social Impacts of AI","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian AIDS Treatment Information Exchange","funders":"","keywords":"Benchmarking; Task (project management); Deliverable; Benchmark (surveying); Index (typography); Trustworthiness; Troubleshooting","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04463449,0.0010323,0.0007040919,0.0114007,0.001679789,0.005403804,0.001584533,0.001699771,0.005644825],"category_scores_gemma":[0.2497424,0.0004061485,0.001388118,0.009353141,0.003056465,0.007032663,0.00463275,0.002301743,0.001853708],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003897073,"about_ca_system_score_gemma":0.003101081,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002007222,"about_ca_topic_score_gemma":0.00170928,"domain_scores_codex":[0.9262466,0.03342474,0.006719219,0.002792625,0.02977662,0.001040138],"domain_scores_gemma":[0.8340729,0.08420088,0.01958533,0.01910738,0.03990766,0.00312579],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004545542,0.0005319865,0.06846623,0.002342174,0.000487591,0.0001259125,0.01235468,0.008876099,0.003628134,0.3385177,0.02920235,0.5350125],"study_design_scores_gemma":[0.0001658704,0.002944747,0.24881,0.003789317,0.000329921,0.000680121,0.01019139,0.07083525,0.0064541,0.4461162,0.2092348,0.0004482995],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.173052,0.007496073,0.6663218,0.005626338,0.001292582,0.004922005,0.003914373,0.002189079,0.1351857],"genre_scores_gemma":[0.6902708,0.001547117,0.288952,0.0004173198,0.0004040784,0.007103087,0.004258015,0.0006983031,0.006349288],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9553655,"threshold_uncertainty_score":0.2360526,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08604458457957553,"score_gpt":0.3455447881666049,"score_spread":0.2595002035870294,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}