{"id":"W4385572760","doi":"10.18653/v1/2022.emnlp-main.23","title":"Certified Error Control of Candidate Set Pruning for Two-Stage Relevance Ranking","year":2022,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canada First Research Excellence Fund; Compute Canada","keywords":"Pruning; Ranking (information retrieval); Computer science; Relevance (law); Set (abstract data type); Data mining; Domain (mathematical analysis); Error detection and correction; Artificial intelligence; Machine learning; Algorithm; Mathematics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000555432,0.00008429104,0.0001731595,0.00006479098,0.0002050999,0.00003349724,0.0006647062,0.000013888,0.000058301],"category_scores_gemma":[0.00005809893,0.00008293657,0.00006000441,0.0001582173,0.00001337469,0.0001829239,0.0002438877,0.0001047857,0.000001142867],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004954863,"about_ca_system_score_gemma":0.00007216195,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001931485,"about_ca_topic_score_gemma":0.00004289033,"domain_scores_codex":[0.9988783,0.00006331342,0.0002687373,0.0003048897,0.0002398087,0.0002449783],"domain_scores_gemma":[0.9991207,0.0002025637,0.0001335998,0.0004556769,0.00005026894,0.00003722373],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001614886,0.00006281675,0.0009685694,0.0001199579,0.00007585067,0.00001793873,0.004840313,0.5662012,0.03983111,0.3550464,0.0007178403,0.03195653],"study_design_scores_gemma":[0.001405022,0.0000477738,0.00002139132,0.00000698936,0.000004986355,0.000003207042,0.0001119884,0.9875099,0.002709756,0.001286167,0.00678377,0.0001090464],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03553933,0.00006444578,0.9620025,0.0008611033,0.0003067071,0.0003383486,0.00001911883,0.0001007858,0.0007677299],"genre_scores_gemma":[0.9493737,5.335548e-7,0.04865155,0.0005086305,0.00002606294,0.00008029269,0.000002743581,0.000008216163,0.001348268],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9138344,"threshold_uncertainty_score":0.3382055,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04368542010145735,"score_gpt":0.2910072447144538,"score_spread":0.2473218246129965,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}