{"id":"W6966350018","doi":"10.48448/2fkc-k105","title":"ECBD: Evidence-Centered Benchmark Design for NLP","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Benchmark (surveying); Benchmarking; Process (computing); Measure (data warehouse); Documentation","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1681353,0.00176377,0.001562518,0.0108273,0.002554036,0.0134639,0.005682678,0.003840048,0.0126976],"category_scores_gemma":[0.3913643,0.001723305,0.001986879,0.008274759,0.005776943,0.009844277,0.01412851,0.004764027,0.003010384],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.009366308,"about_ca_system_score_gemma":0.03052748,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005569262,"about_ca_topic_score_gemma":0.00782425,"domain_scores_codex":[0.8076257,0.1355411,0.0247449,0.00690536,0.02336957,0.001813364],"domain_scores_gemma":[0.6385164,0.2058801,0.02568416,0.04883324,0.07508536,0.006000678],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0007568432,0.0005743574,0.007554749,0.007966585,0.0003678854,0.0003882703,0.005569024,0.02110357,0.002876369,0.3717427,0.02578634,0.5553133],"study_design_scores_gemma":[0.00125184,0.001026084,0.006769325,0.01351772,0.0004352608,0.0004471974,0.004540468,0.07191881,0.01239855,0.5763304,0.3109822,0.0003821666],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.008072738,0.001329616,0.9362714,0.007273241,0.0002610211,0.008803964,0.002344179,0.003926187,0.03171758],"genre_scores_gemma":[0.03621788,0.0003318686,0.9552923,0.0005118653,0.00002070633,0.005217771,0.001137118,0.0002541083,0.00101638],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1681353,"threshold_uncertainty_score":0.889195,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1571169989981255,"score_gpt":0.3803313389014374,"score_spread":0.2232143399033119,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}