{"id":"W2895601233","doi":"10.1145/3209280.3209532","title":"Active High-Recall Information Retrieval from Domain-Specific Text Corpora based on Query Documents","year":2018,"lang":"en","type":"article","venue":"","topic":"Machine Learning and Algorithms","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Information retrieval; Ranking (information retrieval); Domain (mathematical analysis); Document retrieval; Class (philosophy); Artificial intelligence; Precision and recall; Active learning (machine learning); Natural language processing","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004275467,0.001232074,0.002136311,0.005388142,0.001117832,0.00324248,0.00311912,0.001988,0.002478207],"category_scores_gemma":[0.0109116,0.0006169966,0.001171559,0.00368587,0.001153495,0.007131976,0.001727575,0.001522118,0.004420116],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008400793,"about_ca_system_score_gemma":0.0009936198,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001826581,"about_ca_topic_score_gemma":0.002958436,"domain_scores_codex":[0.9966509,0.0009065261,0.0002589216,0.0007746102,0.001258967,0.0001502269],"domain_scores_gemma":[0.9934013,0.003224041,0.0006292089,0.001351446,0.001240376,0.0001536456],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008667642,0.0008136752,0.002725781,0.001198901,0.0002299651,0.0004628185,0.0008624393,0.01407512,0.07925063,0.008344654,0.01447993,0.8766893],"study_design_scores_gemma":[0.0002943463,0.001353422,0.007526494,0.0002238568,0.000502367,0.00283418,0.0008152014,0.7709181,0.1372929,0.02209108,0.05584143,0.0003065766],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03762144,0.003479283,0.9465677,0.0005823289,0.0001310024,0.000395295,0.0004636399,0.007121203,0.00363803],"genre_scores_gemma":[0.3527752,0.002378462,0.6264937,0.0009130462,0.0005568783,0.0006544732,0.004393496,0.0005334482,0.01130135],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005388142,"threshold_uncertainty_score":0.02261114,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.007305110214433513,"score_gpt":0.2241554571873962,"score_spread":0.2168503469729627,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}