{"id":"W4409348492","doi":"10.1609/aaai.v39i27.35108","title":"Certified Trustworthiness in the Era of Large Language Models","year":2025,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Access Control and Trust","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Trustworthiness; Certification; Computer science; Linguistics; Political science; Computer security; Philosophy; Law","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001392497,0.0001362211,0.0002596864,0.00009958183,0.000301627,0.0001108627,0.001537724,0.0001053721,0.0001052482],"category_scores_gemma":[0.0006500309,0.00008486673,0.000109871,0.001034915,0.0004393155,0.0002822143,0.000125406,0.0003804026,0.000007942022],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003145688,"about_ca_system_score_gemma":0.0001809851,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002094928,"about_ca_topic_score_gemma":0.001964226,"domain_scores_codex":[0.9984073,0.00006031264,0.0004849728,0.0002419397,0.000484134,0.0003213211],"domain_scores_gemma":[0.9989135,0.0002062599,0.0002613462,0.0001851352,0.0004030074,0.00003078719],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00009040764,0.0001621397,0.0008935957,0.0000295399,0.00001021986,3.348903e-7,0.02101413,0.00006351894,0.001268171,0.949973,0.0000689191,0.02642605],"study_design_scores_gemma":[0.00014506,0.00007019703,0.002817175,0.0006368665,0.00005550141,3.71231e-7,0.08829442,0.02363515,0.0560025,0.8276325,0.0004229711,0.000287314],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7841916,0.0001399818,0.0009122492,0.01662276,0.0003717004,0.000791101,0.00001339186,0.00003879968,0.1969184],"genre_scores_gemma":[0.998963,0.00006450019,0.00003910099,0.0004232884,0.00005454858,0.00003539033,4.329995e-7,0.000005255811,0.0004144933],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2147714,"threshold_uncertainty_score":0.3460765,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0646789908497029,"score_gpt":0.3533518392017687,"score_spread":0.2886728483520657,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}