{"id":"W2120308175","doi":"10.1016/s0306-4573(00)00010-8","title":"Variations in relevance judgments and the measurement of retrieval effectiveness","year":2000,"lang":"en","type":"article","venue":"Information Processing & Management","topic":"Information Retrieval and Search Behavior","field":"Computer Science","cited_by":507,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"National Institute of Standards and Technology; University of Waterloo","keywords":"Relevance (law); Information retrieval; Computer science; NIST; Set (abstract data type); Test (biology); Reliability (semiconductor); Ranking (information retrieval); Document retrieval; Natural language processing","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02027675,0.0003902094,0.0008037236,0.003208354,0.000516544,0.002878197,0.0007765157,0.001811885,0.001334948],"category_scores_gemma":[0.3084425,0.0004411684,0.0004142646,0.002551426,0.001332262,0.002222596,0.00100395,0.001700244,0.0005639677],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008852497,"about_ca_system_score_gemma":0.0004947886,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001455787,"about_ca_topic_score_gemma":0.001216212,"domain_scores_codex":[0.9669511,0.0202517,0.002204835,0.002753044,0.007238827,0.0006004745],"domain_scores_gemma":[0.6498775,0.3022025,0.02286725,0.01484736,0.00797183,0.002233591],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.009852016,0.002030525,0.6785566,0.0008471128,0.001720816,0.0003162361,0.006507598,0.01066102,0.09339182,0.008429323,0.002665767,0.1850212],"study_design_scores_gemma":[0.0001086026,0.001155992,0.9737563,0.00004207894,0.0002059622,0.0004813164,0.0002902412,0.008780705,0.007849541,0.006406734,0.0008049486,0.0001176381],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9859698,0.001376231,0.007141199,0.0001545675,0.00004093683,0.0001107648,0.000202766,0.0001084405,0.004895355],"genre_scores_gemma":[0.9975782,0.00009921359,0.001760935,0.0000537457,0.0000328627,0.00004654071,0.0001412378,0.00004071946,0.0002466349],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02027675,"threshold_uncertainty_score":0.107235,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01358723951661481,"score_gpt":0.2439882889508641,"score_spread":0.2304010494342493,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}