{"id":"W2333968622","doi":"10.1177/154193120204601408","title":"How Good is Search Engine Ranking? a Validation Study with Human Judges","year":2002,"lang":"en","type":"article","venue":"Proceedings of the Human Factors and Ergonomics Society Annual Meeting","topic":"Information Retrieval and Search Behavior","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Ranking (information retrieval); Relevance (law); Information retrieval; Search engine; Set (abstract data type); Correlation; Psychology; Computer science; Contrast (vision); Statistics; Mathematics; Artificial intelligence; Political science; Law","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04134995,0.000629356,0.0008289657,0.001730678,0.001082155,0.002022499,0.000707927,0.001365483,0.0009531539],"category_scores_gemma":[0.1918674,0.0003340126,0.0006851458,0.001336352,0.001447953,0.001391874,0.0008390341,0.0008379954,0.0006605104],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006008776,"about_ca_system_score_gemma":0.0006789393,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002763499,"about_ca_topic_score_gemma":0.004482842,"domain_scores_codex":[0.9597747,0.02849477,0.002100081,0.002558693,0.006395958,0.0006758427],"domain_scores_gemma":[0.7532694,0.1872912,0.01090605,0.01318378,0.03340481,0.00194462],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.006470553,0.003401393,0.6271946,0.00187033,0.001819689,0.000730839,0.02880744,0.01162578,0.04028446,0.002659602,0.01178178,0.2633536],"study_design_scores_gemma":[0.0008653245,0.01459641,0.849393,0.0004939266,0.001135085,0.002017894,0.01157757,0.06425755,0.03206114,0.004319855,0.01873061,0.0005516812],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9776558,0.001086614,0.01300458,0.0003118715,0.0001670513,0.0003129056,0.0002950316,0.0001379341,0.00702824],"genre_scores_gemma":[0.9893479,0.0003047748,0.008672385,0.000136886,0.00009300428,0.000137962,0.0003489244,0.00004535459,0.0009127216],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04134995,"threshold_uncertainty_score":0.2186821,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02802410079066921,"score_gpt":0.2392387341369389,"score_spread":0.2112146333462697,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}