{"id":"W4393090174","doi":"10.1007/978-3-031-56069-9_20","title":"eval-rationales: An End-to-End Toolkit to Explain and Evaluate Transformers-Based Models","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"Fonds de Recherche du Québec - Santé","keywords":"Interpretability; Computer science; Transformer; Artificial intelligence; End-to-end principle; Machine learning; Data science; Software engineering; Engineering","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.002348856,0.0007324776,0.000586314,0.001688896,0.0004001659,0.001583486,0.003001995,0.0002963758,0.00007125084],"category_scores_gemma":[0.0001013532,0.0006900328,0.0001380535,0.001275418,0.0004974104,0.001749873,0.0007453251,0.0007089979,0.0002474384],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005054542,"about_ca_system_score_gemma":0.001164037,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00008898941,"about_ca_topic_score_gemma":0.000687808,"domain_scores_codex":[0.993681,0.00007415668,0.0007463442,0.002650059,0.001864157,0.0009843165],"domain_scores_gemma":[0.9968401,0.0006207649,0.0001173248,0.001337446,0.0004521378,0.0006322192],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001842992,0.00002986155,0.000002302177,0.00005259132,0.00001061464,0.00008591769,0.004387092,0.482115,0.0009180076,0.09526249,0.00002002572,0.4170977],"study_design_scores_gemma":[0.00008019485,0.0004559434,0.00000519166,0.0003482302,0.00001219535,0.00002488446,0.000002720725,0.7210269,0.006297399,0.269944,0.001153638,0.0006486516],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001011865,0.000361552,0.9883252,0.004497538,0.001276138,0.001103763,0.00002618791,0.0002505907,0.003147114],"genre_scores_gemma":[0.4843856,0.00003409334,0.5064335,0.007440734,0.0005467195,0.0001166543,0.00001755614,0.00009818701,0.0009268456],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.4833738,"threshold_uncertainty_score":0.9995551,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04916689337903857,"score_gpt":0.3026271434946071,"score_spread":0.2534602501155685,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}