{"id":"W4409916786","doi":"10.1109/tse.2025.3565387","title":"Question Selection for Multimodal Code Search Synthesis Using Probabilistic Version Spaces","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"National Natural Science Foundation of China","keywords":"Computer science; Selection (genetic algorithm); Probabilistic logic; Modal; Programming language; Code (set theory); Theoretical computer science; Artificial intelligence; Set (abstract data type)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00431377,0.001236836,0.001291846,0.002160809,0.0009557842,0.00273508,0.002263417,0.001900168,0.01964309],"category_scores_gemma":[0.02398555,0.0008037565,0.001817168,0.0008878155,0.001510108,0.004321248,0.004903532,0.001232394,0.003117298],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001250636,"about_ca_system_score_gemma":0.001159449,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001660631,"about_ca_topic_score_gemma":0.001843963,"domain_scores_codex":[0.9939803,0.002643904,0.0004342015,0.001328043,0.001341507,0.000272014],"domain_scores_gemma":[0.9901937,0.00741276,0.000396851,0.001020389,0.0007538295,0.0002224585],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001312545,0.0002567356,0.002356506,0.0008732995,0.0001293411,0.0007282782,0.002793019,0.1297603,0.0302563,0.1727809,0.009061106,0.6496918],"study_design_scores_gemma":[0.0000873324,0.0001413765,0.0002788681,0.00006746634,0.00004258121,0.0002051137,0.000274333,0.8888962,0.01634489,0.08181558,0.01178979,0.00005655768],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01078015,0.0001320116,0.9775885,0.0001886769,0.00003236389,0.0001714453,0.0002073731,0.008015975,0.002883599],"genre_scores_gemma":[0.3096611,0.0001297694,0.6815495,0.0001919406,0.00004816418,0.0005934131,0.0008937071,0.001492215,0.005440101],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01964309,"threshold_uncertainty_score":0.06571269,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01307841608488208,"score_gpt":0.2768731439809722,"score_spread":0.2637947278960901,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}