{"id":"W4409671330","doi":"10.1145/3696410.3714701","title":"MixedSAND: Semantic Annotation of Mixed-unit Numeric Data","year":2025,"lang":"en","type":"article","venue":"","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Annotation; Natural language processing; Unit (ring theory); Information retrieval; Semantic annotation; Artificial intelligence; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003229577,0.001540626,0.0009628843,0.006107355,0.001421735,0.003582581,0.002608999,0.001739919,0.006472299],"category_scores_gemma":[0.01655246,0.0006867766,0.001475825,0.006890098,0.001240965,0.007344471,0.005311616,0.002002822,0.003133277],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001495004,"about_ca_system_score_gemma":0.002947412,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009156284,"about_ca_topic_score_gemma":0.01894069,"domain_scores_codex":[0.9965237,0.0007240897,0.0004047526,0.001044609,0.001143199,0.0001596984],"domain_scores_gemma":[0.990508,0.003757484,0.0007622601,0.003169578,0.001525319,0.0002772987],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001131639,0.0004444322,0.02149591,0.003946968,0.0003827012,0.001563596,0.005721229,0.02271555,0.03484945,0.08706314,0.2051061,0.6155792],"study_design_scores_gemma":[0.0001042412,0.0001932376,0.009738905,0.0008874597,0.000251987,0.001508432,0.00384314,0.281108,0.06037303,0.1222768,0.5194798,0.000234912],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03056592,0.0008577437,0.8550687,0.001409319,0.0003372441,0.0005734303,0.04773431,0.05269708,0.01075614],"genre_scores_gemma":[0.1082314,0.0005147113,0.8094949,0.0006613148,0.00007790331,0.0004403151,0.07304957,0.00302899,0.00450093],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009156284,"threshold_uncertainty_score":0.02165198,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3282300458693544,"score_gpt":0.491568531296563,"score_spread":0.1633384854272086,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}