{"id":"W7082245767","doi":"10.48448/x0gg-7n40","title":"NativQA: Multilingual Culturally-Aligned Natural Query for LLMs","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Benchmarking; Benchmark (surveying); Construct (python library); Scripting language; Question answering; Natural language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004895071,0.002492441,0.001257959,0.003197637,0.001307069,0.002202447,0.004076873,0.001862467,0.01048278],"category_scores_gemma":[0.01813545,0.0006908505,0.002116995,0.002250425,0.001120136,0.004759191,0.005929495,0.002439123,0.008587503],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002278821,"about_ca_system_score_gemma":0.003218754,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03053126,"about_ca_topic_score_gemma":0.03506932,"domain_scores_codex":[0.9939559,0.002214714,0.0006189273,0.001722986,0.001125548,0.0003618564],"domain_scores_gemma":[0.9940779,0.001985336,0.0002845416,0.001918482,0.001301839,0.0004319036],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001555373,0.0007602282,0.01231023,0.003746422,0.000485471,0.0006828782,0.002361214,0.02671408,0.02502933,0.02131114,0.7553539,0.1496896],"study_design_scores_gemma":[0.000804673,0.0005485219,0.01487145,0.0004201372,0.0001819198,0.001017777,0.002720686,0.3479287,0.03398015,0.04253152,0.5545645,0.0004300221],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"methods","genre_scores_codex":[0.06372099,0.002948748,0.2317267,0.003100507,0.0008575279,0.002821851,0.3829027,0.2895105,0.02241056],"genre_scores_gemma":[0.1303284,0.0004039571,0.2188133,0.001459706,0.0001037108,0.001963262,0.6367985,0.005756863,0.004372259],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.03053126,"threshold_uncertainty_score":0.06070709,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01320524591029865,"score_gpt":0.2839877655436504,"score_spread":0.2707825196333517,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}