{"id":"W2806710540","doi":"10.18653/v1/n18-4004","title":"A Generalized Knowledge Hunting Framework for the Winograd Schema Challenge","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Schema (genetic algorithms); Computer science; Theoretical computer science; Artificial intelligence; Knowledge management; Algebra over a field; Mathematics; Machine learning; Pure mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005746372,0.0009360344,0.00157395,0.003649595,0.002585605,0.006690883,0.00681315,0.003085487,0.01511356],"category_scores_gemma":[0.01576877,0.0009071537,0.002426535,0.004977542,0.002017436,0.01464204,0.01210036,0.004914144,0.005541607],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001524423,"about_ca_system_score_gemma":0.003704047,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00982092,"about_ca_topic_score_gemma":0.02338065,"domain_scores_codex":[0.9953465,0.001598868,0.0004109261,0.0009414376,0.001391596,0.0003107674],"domain_scores_gemma":[0.9946982,0.001717803,0.0001884995,0.002230003,0.0007753016,0.0003901936],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005188188,0.0005055636,0.002258502,0.0008602372,0.0003308303,0.0005495577,0.0008549986,0.02131123,0.00198859,0.4017054,0.1823207,0.3867956],"study_design_scores_gemma":[0.0001429172,0.00007202974,0.0003506376,0.0002687092,0.00009753145,0.0004293942,0.0007211903,0.1960206,0.002306738,0.6645983,0.134938,0.00005399632],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01336823,0.0041738,0.9339215,0.008200857,0.0004832312,0.0007683037,0.007930246,0.0121377,0.01901612],"genre_scores_gemma":[0.1089359,0.001648654,0.8545424,0.001786544,0.0002582429,0.0004193356,0.02149431,0.001554388,0.009360127],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01511356,"threshold_uncertainty_score":0.05055988,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03997580010602558,"score_gpt":0.3405327431500678,"score_spread":0.3005569430440422,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}