{"id":"W4414459164","doi":"10.1109/acdsa65407.2025.11166412","title":"How Reliable Is Semantic Search in Industrial Computing Domain? A Statistical Evaluation Pipeline","year":2025,"lang":"en","type":"article","venue":"","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Advantech AMT (Canada)","funders":"","keywords":"Set (abstract data type); Pipeline (software); Semantic search; Cover (algebra); Key (lock); Range (aeronautics); Semantic data model; Ground truth","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0566342,0.001555511,0.001918665,0.0135476,0.001381765,0.007175527,0.002231299,0.002397474,0.003780279],"category_scores_gemma":[0.2454204,0.0004928892,0.001299226,0.009106664,0.002476248,0.01376881,0.003669544,0.001871559,0.002871242],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002189919,"about_ca_system_score_gemma":0.004192438,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005086842,"about_ca_topic_score_gemma":0.006245792,"domain_scores_codex":[0.9268167,0.0384268,0.006284926,0.004585749,0.02223923,0.001646515],"domain_scores_gemma":[0.7944503,0.1396437,0.008515594,0.01949706,0.03577461,0.002118739],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00225973,0.0009462389,0.0955827,0.004000399,0.0009305726,0.0004104618,0.002504458,0.06526145,0.01852022,0.03149102,0.03314279,0.74495],"study_design_scores_gemma":[0.0002564757,0.001652117,0.04726784,0.001240131,0.0004283425,0.0006933778,0.004048584,0.7803485,0.04454389,0.08454353,0.03460828,0.0003688162],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3043323,0.007125017,0.6218396,0.005531884,0.0004300065,0.00187571,0.006714345,0.0133802,0.03877098],"genre_scores_gemma":[0.8184137,0.001219274,0.1681888,0.0003920933,0.0001450235,0.0007034262,0.008001298,0.00120273,0.001733766],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0566342,"threshold_uncertainty_score":0.2995139,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06602572388845862,"score_gpt":0.3357651664666445,"score_spread":0.2697394425781859,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}