{"id":"W4385079050","doi":"10.1145/3597465.3605221","title":"A Human-in-the-loop Workflow for Multi-Factorial Sensitivity Analysis of Algorithmic Rankers","year":2023,"lang":"en","type":"article","venue":"","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"York University; National Science Foundation","keywords":"Computer science; Leverage (statistics); Workflow; Data exploration; Data mining; Machine learning; Sensitivity (control systems); Visualization; Artificial intelligence; Database","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001916287,0.0001404804,0.0003378655,0.0006299813,0.0001489467,0.0001033624,0.0006678312,0.00007500155,0.00001722775],"category_scores_gemma":[0.0002136252,0.000112722,0.0002791196,0.004600821,0.0000647354,0.0003101587,0.0001261864,0.0001059879,0.00004122443],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004631812,"about_ca_system_score_gemma":0.00004529555,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009396141,"about_ca_topic_score_gemma":0.00287716,"domain_scores_codex":[0.9982307,0.0001914011,0.0003881224,0.0004287881,0.0003403414,0.0004206715],"domain_scores_gemma":[0.9983345,0.0007089192,0.0001020706,0.0006564783,0.0001487087,0.00004932393],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002214343,0.002235687,0.01737933,0.0002012647,0.003283686,0.0006619336,0.04978581,0.2178934,0.1163232,0.339239,0.0130493,0.2397259],"study_design_scores_gemma":[0.0002554519,0.00005618713,0.004377255,0.000006915584,0.00007757635,7.377881e-7,0.0005489605,0.9822116,0.01048128,0.001624252,0.0001952123,0.0001646095],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.135786,0.000006077057,0.8625309,0.0004155315,0.0004488605,0.0004593086,0.000009163573,0.0001549435,0.0001892741],"genre_scores_gemma":[0.9697137,0.000003255505,0.02946484,0.0001349708,0.0001069312,0.0000566291,0.00001090415,0.000009173032,0.0004995289],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8339278,"threshold_uncertainty_score":0.4596669,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09679615228762019,"score_gpt":0.3597003880135416,"score_spread":0.2629042357259215,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}