{"id":"W4381329270","doi":"10.1145/3589322","title":"Maestro: Automatic Generation of Comprehensive Benchmarks for Question Answering Over Knowledge Graphs","year":2023,"lang":"en","type":"article","venue":"Proceedings of the ACM on Management of Data","topic":"Advanced Graph Neural Networks","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Benchmark (surveying); Question answering; Knowledge base; Vocabulary; Natural language; Information retrieval; Set (abstract data type); Natural language understanding; Knowledge graph; Usability; Artificial intelligence; Natural language processing; Programming language; Human–computer interaction; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004043825,0.002868759,0.0009871292,0.005460505,0.000757851,0.002009164,0.003242787,0.001710442,0.009022462],"category_scores_gemma":[0.03305459,0.000874062,0.001540313,0.002840475,0.0007950977,0.003001204,0.00316305,0.00178496,0.004532638],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001360081,"about_ca_system_score_gemma":0.001837604,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004890708,"about_ca_topic_score_gemma":0.006124322,"domain_scores_codex":[0.9955574,0.0016311,0.0004574482,0.001028714,0.00107599,0.0002494612],"domain_scores_gemma":[0.98413,0.009549213,0.0007482918,0.001918249,0.003216448,0.0004378261],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001331067,0.001064848,0.009405471,0.004960965,0.0004104489,0.001345342,0.0025553,0.05863292,0.04068023,0.02131938,0.3017087,0.5565853],"study_design_scores_gemma":[0.0007349434,0.0007182924,0.008176192,0.0004824784,0.0001674592,0.0007170358,0.001384482,0.6931043,0.06720524,0.03845834,0.1886228,0.0002284457],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08346705,0.002578621,0.4931529,0.001152231,0.0008159296,0.002927177,0.06082034,0.3354835,0.01960238],"genre_scores_gemma":[0.1786785,0.0007028808,0.5897623,0.0004868461,0.0001009741,0.003137813,0.2076795,0.01448213,0.004969027],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009022462,"threshold_uncertainty_score":0.0301832,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1019414078019133,"score_gpt":0.3440120252621289,"score_spread":0.2420706174602156,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}