{"id":"W4404315596","doi":"10.1115/detc2024-139024","title":"DesignQA: Benchmarking Multimodal Large Language Models on Questions Grounded in Engineering Documentation","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Autodesk (Canada)","funders":"","keywords":"Documentation; Benchmarking; Computer science; Software engineering; Natural language processing; Artificial intelligence; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008252601,0.002047981,0.0007596289,0.002704026,0.0007235479,0.003175122,0.00284633,0.002660615,0.01099687],"category_scores_gemma":[0.04999628,0.0005178452,0.001352611,0.001539003,0.001172099,0.003316794,0.004073734,0.00238334,0.005656939],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001805184,"about_ca_system_score_gemma":0.001955543,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007070434,"about_ca_topic_score_gemma":0.007100747,"domain_scores_codex":[0.9854053,0.009600068,0.001058984,0.001619024,0.001992593,0.0003240502],"domain_scores_gemma":[0.9610438,0.0282142,0.00108437,0.004864017,0.003958797,0.0008348047],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002534398,0.004422948,0.01804176,0.005279575,0.0006456717,0.0007740837,0.004998615,0.2039376,0.02621124,0.02044864,0.1365529,0.5761526],"study_design_scores_gemma":[0.0007780142,0.002039304,0.009350726,0.0005819014,0.0001460339,0.0005415877,0.002464642,0.8344962,0.03206767,0.02582834,0.09145207,0.0002535535],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4924162,0.003600102,0.3355539,0.002645831,0.0009188164,0.002786068,0.03569705,0.08319153,0.04319035],"genre_scores_gemma":[0.6730596,0.000554099,0.2285219,0.001057262,0.00011558,0.002193948,0.08337111,0.002911471,0.008215001],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01099687,"threshold_uncertainty_score":0.04364449,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01049637198367985,"score_gpt":0.28531260999824,"score_spread":0.2748162380145602,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}