{"id":"W4411638719","doi":"10.18653/v1/2024.eacl-tutorials.2","title":"Item Response Theory for Natural Language Processing","year":2024,"lang":"en","type":"article","venue":"","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Atomic Energy of Canada Limited; York University; Institute for Catastrophic Loss Reduction; University of Notre Dame","keywords":"Computer science; Natural (archaeology); Natural language processing","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0754396,0.004319437,0.005978222,0.00896003,0.002113385,0.006146647,0.00582661,0.004302719,0.01221921],"category_scores_gemma":[0.2092671,0.003218111,0.005794351,0.01003429,0.006838587,0.008716978,0.004380741,0.007899455,0.005913902],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004488585,"about_ca_system_score_gemma":0.002662058,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004433892,"about_ca_topic_score_gemma":0.002429499,"domain_scores_codex":[0.8932576,0.09412817,0.002703336,0.003613398,0.005649026,0.0006485498],"domain_scores_gemma":[0.7860878,0.1888204,0.002937513,0.01308514,0.00847325,0.0005958552],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004868339,0.0008323899,0.008120402,0.002847456,0.002869937,0.0003433737,0.003275581,0.02957392,0.0006858353,0.6108443,0.03811377,0.3020063],"study_design_scores_gemma":[0.0002539064,0.0001458098,0.002862642,0.0004011085,0.0002015486,0.0001327967,0.0003787656,0.04344134,0.0002149431,0.943498,0.00837362,0.00009565688],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.005676762,0.00905881,0.9688051,0.003438577,0.000706931,0.00169004,0.00103252,0.001434344,0.008157078],"genre_scores_gemma":[0.1530573,0.003951556,0.8218957,0.001815973,0.001277584,0.01177298,0.002433887,0.0006920674,0.003103113],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0754396,"threshold_uncertainty_score":0.3989675,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4064670903822485,"score_gpt":0.5454917346558178,"score_spread":0.1390246442735693,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}