{"id":"W2576107865","doi":"","title":"WaterlooClarke: TREC 2015 LiveQA Track","year":2015,"lang":"en","type":"article","venue":"Text REtrieval Conference","topic":"Expert finding and Q&A systems","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Question answering; Computer science; NIST; Track (disk drive); Task (project management); Questions and answers; Information retrieval; World Wide Web; Natural language processing; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01767681,0.003275532,0.002239976,0.005601061,0.005755822,0.006211665,0.004462972,0.004319162,0.06229487],"category_scores_gemma":[0.02038388,0.001340707,0.001259041,0.003499888,0.002043619,0.006834551,0.004203525,0.004520084,0.04779785],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01184978,"about_ca_system_score_gemma":0.01215402,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.2729172,"about_ca_topic_score_gemma":0.4138702,"domain_scores_codex":[0.9894713,0.003032071,0.0006642498,0.002043856,0.003711556,0.001076999],"domain_scores_gemma":[0.9723566,0.004154014,0.0006544957,0.003500186,0.01643583,0.00289891],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.000244418,0.0002902453,0.0003695688,0.0003773441,0.00004819812,0.00004651442,0.0001707946,0.0004100239,0.002821389,0.0005639988,0.9795702,0.01508722],"study_design_scores_gemma":[0.001312188,0.0006230748,0.01439655,0.0003222192,0.0001135889,0.0002899723,0.0008670827,0.01662953,0.0176207,0.00388512,0.9436982,0.0002416955],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.05063422,0.008167766,0.03970778,0.01559586,0.007181663,0.01009581,0.6429985,0.07643976,0.1491786],"genre_scores_gemma":[0.04870628,0.0006194817,0.02694652,0.002673458,0.0006146245,0.00234178,0.8348454,0.003932325,0.07932016],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.2729172,"threshold_uncertainty_score":0.5426574,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07404688050771699,"score_gpt":0.2936623825454258,"score_spread":0.2196155020377088,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}