{"id":"W4385079050","doi":"10.1145/3597465.3605221","title":"A Human-in-the-loop Workflow for Multi-Factorial Sensitivity Analysis of Algorithmic Rankers","year":2023,"lang":"en","type":"article","venue":"","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"York University; National Science Foundation","keywords":"Computer science; Leverage (statistics); Workflow; Data exploration; Data mining; Machine learning; Sensitivity (control systems); Visualization; Artificial intelligence; Database","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01531139,0.002787918,0.001491362,0.002851806,0.00141278,0.004074247,0.003161669,0.001985765,0.03553225],"category_scores_gemma":[0.05034798,0.001221598,0.002950603,0.001130258,0.001670852,0.003013529,0.004675643,0.00331712,0.004075329],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00220258,"about_ca_system_score_gemma":0.003569995,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002962922,"about_ca_topic_score_gemma":0.003896231,"domain_scores_codex":[0.9940078,0.003225938,0.0004128791,0.0009385436,0.001163053,0.0002518551],"domain_scores_gemma":[0.9560329,0.0350715,0.00159935,0.004295866,0.002412789,0.0005876144],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00130671,0.0009795746,0.00771842,0.001501599,0.0004144784,0.001496109,0.005527526,0.2856818,0.01413439,0.1820912,0.03007978,0.4690683],"study_design_scores_gemma":[0.0002430379,0.0001155264,0.0005603479,0.0001537485,0.00004170003,0.0001289418,0.0002903445,0.7304159,0.006786805,0.2434254,0.01774566,0.00009260463],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002594175,0.00003836473,0.9817017,0.0002937586,0.00003216038,0.000367345,0.0004296863,0.01276796,0.001774748],"genre_scores_gemma":[0.07047956,0.00007171497,0.9236999,0.0001846446,0.0000305719,0.001279585,0.0008141152,0.001830792,0.001609205],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03553225,"threshold_uncertainty_score":0.1188673,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09679615228762019,"score_gpt":0.3597003880135416,"score_spread":0.2629042357259215,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}