{"id":"W6931734854","doi":"10.5281/zenodo.8267719","title":"Artifacts for Resilience Assessment of Large Language Models under Transient Hardware Faults","year":2023,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Prenatal Screening and Diagnostics","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Transient (computer programming); Resilience (materials science); Language model; Software; Simulation language; Key (lock)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002466052,0.002177588,0.00061308,0.002281139,0.0006397948,0.00235107,0.002114359,0.001764883,0.08390965],"category_scores_gemma":[0.02028371,0.0009768631,0.001833946,0.001213752,0.0006354323,0.002147667,0.002626521,0.002142681,0.03183535],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006590418,"about_ca_system_score_gemma":0.001487531,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002849902,"about_ca_topic_score_gemma":0.003573009,"domain_scores_codex":[0.9967901,0.0007412366,0.0003061055,0.0003018816,0.001681243,0.0001793839],"domain_scores_gemma":[0.9889565,0.004372665,0.0004398458,0.003760371,0.002224631,0.0002459399],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0008765336,0.0003564756,0.003042585,0.002582101,0.0001845253,0.001252797,0.0007328875,0.06197995,0.02088504,0.08270075,0.5473385,0.2780679],"study_design_scores_gemma":[0.0002873556,0.0001800767,0.002288726,0.0006354415,0.00009726498,0.001323976,0.0001601581,0.2141569,0.04717216,0.1000115,0.6335084,0.000178091],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.00573711,0.0002962049,0.7513956,0.0007230111,0.000454575,0.0005504424,0.0510701,0.1545453,0.03522778],"genre_scores_gemma":[0.1285979,0.0007679093,0.5972349,0.0009221704,0.000306213,0.001835961,0.15793,0.06706257,0.04534234],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.08390965,"threshold_uncertainty_score":0.2807057,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04434567615054684,"score_gpt":0.318185431831423,"score_spread":0.2738397556808762,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}