{"id":"W4417287532","doi":"10.2139/ssrn.5912869","title":"LLMs as Judges: Toward The Automatic Review of GSN-compliant Assurance Cases","year":2025,"lang":"","type":"preprint","venue":"SSRN Electronic Journal","topic":"Safety Systems Engineering in Autonomy","field":"Engineering","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"York University; University of Ottawa","funders":"","keywords":"Leverage (statistics); Quality assurance; Harm; Task (project management); Obstacle; Risk assessment","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1112318,0.0007026618,0.001346583,0.01404597,0.00381787,0.01195634,0.004969743,0.00443823,0.007831331],"category_scores_gemma":[0.3865941,0.000968972,0.0007484105,0.004195579,0.003686383,0.008157576,0.009970915,0.004199302,0.004048687],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004304762,"about_ca_system_score_gemma":0.01496212,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004903483,"about_ca_topic_score_gemma":0.0113623,"domain_scores_codex":[0.845852,0.08492657,0.0105954,0.01214966,0.04262966,0.003846785],"domain_scores_gemma":[0.5899917,0.2049393,0.03077214,0.04869081,0.1168266,0.008779476],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001234106,0.0003864852,0.04813913,0.001346106,0.0002060574,0.001893676,0.01949863,0.007779734,0.02404281,0.1208229,0.1110188,0.6636317],"study_design_scores_gemma":[0.000626638,0.0005117105,0.04095618,0.002500927,0.0004000372,0.001384997,0.01546631,0.2950977,0.05690448,0.2146464,0.3710176,0.0004870796],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1947212,0.0033022,0.6599052,0.03068561,0.001816909,0.003593141,0.001941293,0.008513338,0.09552117],"genre_scores_gemma":[0.6740267,0.0006595582,0.3077126,0.003009843,0.0009263575,0.0009832787,0.001675725,0.0007437333,0.01026225],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8887682,"threshold_uncertainty_score":0.588257,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01453619147121232,"score_gpt":0.2614246803901264,"score_spread":0.2468884889189141,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}