{"id":"W4398757454","doi":"10.1162/tacl_a_00667","title":"Evaluating Correctness and Faithfulness of Instruction-Following Models for Question Answering","year":2024,"lang":"en","type":"article","venue":"Transactions of the Association for Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":69,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Institute for Advanced Research; McGill University; Mila - Quebec Artificial Intelligence Institute; Minnow Environmental (Canada); Research Canada","funders":"","keywords":"Correctness; Computer science; Question answering; Natural language processing; Artificial intelligence; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01909782,0.001115501,0.0009150107,0.001752439,0.0006676409,0.002926078,0.002302674,0.002493935,0.002085758],"category_scores_gemma":[0.1040018,0.0005551682,0.0008927653,0.0009017299,0.001645144,0.003426405,0.002425339,0.002598426,0.0009658962],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002061284,"about_ca_system_score_gemma":0.001676693,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008827999,"about_ca_topic_score_gemma":0.008541726,"domain_scores_codex":[0.9879096,0.007259812,0.0008078537,0.002161837,0.001418576,0.0004422038],"domain_scores_gemma":[0.8792539,0.09748355,0.004336442,0.01080103,0.006468599,0.001656582],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00448744,0.001165706,0.06192508,0.001209271,0.0006914596,0.00030912,0.00464162,0.5150812,0.02795876,0.009305319,0.01128441,0.3619407],"study_design_scores_gemma":[0.00004629346,0.0002981313,0.00293184,0.00004250773,0.00005636044,0.00006380582,0.0001859486,0.9819347,0.009088089,0.004488131,0.0008302592,0.0000339819],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6171094,0.001053758,0.3619538,0.001319619,0.0001916272,0.0006637375,0.001097423,0.01213438,0.004476221],"genre_scores_gemma":[0.9459374,0.00008363485,0.05154453,0.0001741629,0.00002712459,0.0001564039,0.00108523,0.0003100999,0.0006813122],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9809022,"threshold_uncertainty_score":0.1010001,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03778902820660011,"score_gpt":0.3285510699647791,"score_spread":0.290762041758179,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}