{"id":"W4409605519","doi":"10.2196/71687","title":"Detecting Redundant Health Survey Questions by Using Language-Agnostic Bidirectional Encoder Representations From Transformers Sentence Embedding: Algorithm Development Study","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Computer science; Embedding; Sentence; Natural language processing; Artificial intelligence; Data science; World Wide Web","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002494035,0.001445954,0.001064242,0.001319591,0.0005209426,0.001225219,0.001760683,0.001444993,0.003900858],"category_scores_gemma":[0.008799331,0.0004304486,0.0009825446,0.0009795658,0.0005314226,0.002183651,0.001608819,0.001849156,0.001947623],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009679456,"about_ca_system_score_gemma":0.002219072,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00548178,"about_ca_topic_score_gemma":0.006585269,"domain_scores_codex":[0.9985915,0.0004629154,0.0001398077,0.0004324467,0.0002322011,0.0001411843],"domain_scores_gemma":[0.9963709,0.002267184,0.0002247327,0.0002916737,0.0007299423,0.0001154675],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0004963921,0.00040433,0.006501311,0.0003275106,0.0001674558,0.0003349713,0.0003580518,0.05495647,0.01097119,0.003943852,0.01042736,0.9111111],"study_design_scores_gemma":[0.00004935539,0.0001196652,0.000839762,0.00002633563,0.00005548974,0.0001603647,0.0001587017,0.9895704,0.004435722,0.00282416,0.001746195,0.00001392204],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.09163181,0.0009066992,0.8971043,0.0006887756,0.0001914879,0.0006013443,0.0008776479,0.006290529,0.001707381],"genre_scores_gemma":[0.3993621,0.000471428,0.5870333,0.0005008743,0.0001579071,0.0008820177,0.006856823,0.0002985753,0.004436999],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.00548178,"threshold_uncertainty_score":0.01318991,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03442585472957464,"score_gpt":0.3716547472181632,"score_spread":0.3372288924885885,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}