{"id":"W6958058837","doi":"10.60692/mfwkg-4eb72","title":"QRelScore: Better Evaluating Generated Questions with Deeper Understanding of Context-aware Relevance","year":2022,"lang":"en","type":"article","venue":"Greater South Information System","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; Canadian Institute for Advanced Research","funders":"","keywords":"Relevance (law); Adversarial system; Matching (statistics); Context (archaeology); Metric (unit); Rendering (computer graphics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007915342,0.001875095,0.00105005,0.003691927,0.0004989571,0.002249749,0.001237513,0.002021905,0.004750432],"category_scores_gemma":[0.05325777,0.0002376516,0.0007196791,0.001186609,0.0006821225,0.003306251,0.002297543,0.001397175,0.001521244],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00093897,"about_ca_system_score_gemma":0.00116947,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002213994,"about_ca_topic_score_gemma":0.003573897,"domain_scores_codex":[0.9920644,0.003794085,0.0005638251,0.001441612,0.001900898,0.0002351794],"domain_scores_gemma":[0.9729672,0.02024862,0.001516998,0.002320073,0.002304799,0.0006423013],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002172849,0.0008096041,0.0286116,0.002926109,0.0006054227,0.0003839027,0.001552384,0.08313292,0.04982031,0.009973132,0.03137075,0.7886411],"study_design_scores_gemma":[0.0002609295,0.001698298,0.02630575,0.0002223222,0.0002637353,0.0007093057,0.0005152348,0.8778665,0.05041488,0.01873014,0.02275311,0.0002598104],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2880936,0.006832324,0.6541147,0.0008359047,0.0006039947,0.001738069,0.005483459,0.02599891,0.016299],"genre_scores_gemma":[0.7266391,0.0004556022,0.2598216,0.0003974864,0.0001535505,0.0005592172,0.007556923,0.0007998645,0.003616724],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007915342,"threshold_uncertainty_score":0.04186088,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0923810007886917,"score_gpt":0.2493279509683739,"score_spread":0.1569469501796822,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}