{"id":"W4409880954","doi":"10.1007/s10994-025-06777-2","title":"Prioritizing the essential: A robust evaluation framework for novelty detection","year":2025,"lang":"en","type":"article","venue":"Machine Learning","topic":"Data Stream Mining Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Novelty; Novelty detection; Computer science; Artificial intelligence; Machine learning; Psychology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03114212,0.003008689,0.003454818,0.005759302,0.00149867,0.006450091,0.004779466,0.003427279,0.002795009],"category_scores_gemma":[0.07182389,0.0007521646,0.001885725,0.002450311,0.002095234,0.00748747,0.006228546,0.004088541,0.0009863948],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002007255,"about_ca_system_score_gemma":0.004376266,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002446958,"about_ca_topic_score_gemma":0.003047774,"domain_scores_codex":[0.9790958,0.006054903,0.001858757,0.002863291,0.009004175,0.001123151],"domain_scores_gemma":[0.9526952,0.02155027,0.003558013,0.004644008,0.01529379,0.002258699],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002705026,0.000910803,0.01670906,0.0008566048,0.0006482695,0.0003280985,0.0003401707,0.08313262,0.0269594,0.06363837,0.0106951,0.7930764],"study_design_scores_gemma":[0.0001017315,0.0008825134,0.002934908,0.000087746,0.0001794111,0.0002768313,0.00008705103,0.9141004,0.01617404,0.06188001,0.00320329,0.0000920416],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01226832,0.0006039861,0.9843945,0.0003287858,0.000101744,0.0002107277,0.0001959607,0.00106218,0.0008338208],"genre_scores_gemma":[0.3956833,0.0003943354,0.5987563,0.0003402822,0.0003755885,0.0004334413,0.0008421829,0.0004894053,0.002685061],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.03114212,"threshold_uncertainty_score":0.1646972,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02530489185627873,"score_gpt":0.3263598317383786,"score_spread":0.3010549398820999,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}