{"id":"W4408184531","doi":"10.1145/3722449.3722461","title":"Report on the 1st Workshop on Large Language Model for Evaluation in Information Retrieval (LLM4Eval 2024) at SIGIR 2024","year":2024,"lang":"en","type":"article","venue":"ACM SIGIR Forum","topic":"Topic Modeling","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"Microsoft (Canada); University of Waterloo","funders":"","keywords":"Information retrieval; Computer science; Natural language processing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.05674778,0.002755925,0.00270539,0.003051185,0.002400398,0.01091567,0.003298965,0.004583383,0.1059989],"category_scores_gemma":[0.05765055,0.00122821,0.002494127,0.002178418,0.001381054,0.01132167,0.01080446,0.008560435,0.07798138],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003446834,"about_ca_system_score_gemma":0.006663677,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01068555,"about_ca_topic_score_gemma":0.01373379,"domain_scores_codex":[0.9754316,0.0119569,0.000888828,0.002567473,0.00765511,0.00150008],"domain_scores_gemma":[0.9404712,0.01961207,0.001157721,0.006593437,0.0235946,0.008570987],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004351033,0.0003351143,0.0005891687,0.0003047637,0.00007327588,0.00007604408,0.0003511653,0.0006996499,0.001619286,0.003642184,0.9115759,0.08029833],"study_design_scores_gemma":[0.0002711882,0.00033775,0.002544988,0.0005359949,0.0001209333,0.0001435896,0.0004587061,0.004171002,0.003484112,0.01268236,0.9751166,0.0001327454],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.02447876,0.06852724,0.3395791,0.1749581,0.1405675,0.006431878,0.05925578,0.01760025,0.1686015],"genre_scores_gemma":[0.06056798,0.01704008,0.1820333,0.02402282,0.02587204,0.006844109,0.1162234,0.01855632,0.54884],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.9432522,"threshold_uncertainty_score":0.3546016,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03542714885077752,"score_gpt":0.3152205595817855,"score_spread":0.279793410731008,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}