{"id":"W179309874","doi":"10.58948/2331-3528.1800","title":"Inconsistent Responsiveness Determination in Document Review: Difference of Opinion or Human Error?","year":2012,"lang":"en","type":"article","venue":"Pace law review","topic":"Legal Systems and Judicial Processes","field":"Social Sciences","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Ambiguity; Computer science; Human error; Quality (philosophy); Information retrieval; Psychology; Data science; Risk analysis (engineering); Epistemology; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.4621453,0.0008368607,0.002481027,0.009135067,0.003299783,0.008428406,0.00396021,0.004035722,0.002395453],"category_scores_gemma":[0.7678267,0.001110734,0.001712461,0.006068004,0.009740588,0.01033176,0.005931162,0.00420284,0.0009584726],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.008522191,"about_ca_system_score_gemma":0.007858316,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002914336,"about_ca_topic_score_gemma":0.003410049,"domain_scores_codex":[0.2901982,0.4997053,0.04806607,0.0270328,0.1307318,0.004265893],"domain_scores_gemma":[0.09446057,0.7707266,0.05231562,0.03221213,0.04900026,0.001284745],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.001300521,0.0004201488,0.08938374,0.005878466,0.003553636,0.001787237,0.09568591,0.006647318,0.007534902,0.1612246,0.03424042,0.5923431],"study_design_scores_gemma":[0.001365845,0.002187407,0.1805046,0.009551043,0.002775077,0.003899028,0.03342911,0.04333993,0.03554024,0.4563482,0.2292317,0.001827874],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2852983,0.02500533,0.5055553,0.09118725,0.005886963,0.003911816,0.0007928489,0.001457847,0.08090447],"genre_scores_gemma":[0.8798699,0.002451146,0.09566595,0.01451393,0.001889263,0.001510671,0.000271195,0.0003368093,0.003491081],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5378547,"threshold_uncertainty_score":0.6632706,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07820519774678054,"score_gpt":0.4173457186269999,"score_spread":0.3391405208802194,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}