{"id":"W2806710540","doi":"10.18653/v1/n18-4004","title":"A Generalized Knowledge Hunting Framework for the Winograd Schema Challenge","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Schema (genetic algorithms); Computer science; Theoretical computer science; Artificial intelligence; Knowledge management; Algebra over a field; Mathematics; Machine learning; Pure mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004229932,0.0001280437,0.0001204981,0.00005582523,0.0003585629,0.0002363164,0.00139931,0.0001059909,0.00002465563],"category_scores_gemma":[0.0002567537,0.00007628193,0.00007933639,0.0003560775,0.00009673464,0.0002847209,0.0003944971,0.0001592177,0.00003442807],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001938133,"about_ca_system_score_gemma":0.00003866147,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001719018,"about_ca_topic_score_gemma":0.00002462312,"domain_scores_codex":[0.999069,0.00002764159,0.0001538271,0.000317936,0.0001272241,0.0003043397],"domain_scores_gemma":[0.9987,0.0003392618,0.00007197153,0.0006309008,0.0002155959,0.00004231208],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000002938757,0.00001943312,0.000009196916,0.00001347772,0.000009815216,4.094837e-7,0.0008544528,2.950679e-8,0.0003629805,0.8348581,0.00147528,0.1623939],"study_design_scores_gemma":[0.0003031968,0.0001886725,0.00002338349,0.0001234055,0.00001433622,0.00001075607,0.00005973768,0.07670539,0.0836036,0.7771633,0.06142814,0.0003761034],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0003858685,0.00630016,0.9853906,0.004334148,0.0003337617,0.0002714451,3.852015e-7,0.0009843615,0.001999272],"genre_scores_gemma":[0.2109826,0.00001788798,0.7873523,0.0006302558,0.0005483273,0.00006449113,3.256096e-7,0.000010798,0.0003930312],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.2105967,"threshold_uncertainty_score":0.3110687,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03997580010602558,"score_gpt":0.3405327431500678,"score_spread":0.3005569430440422,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}