{"id":"W4400653413","doi":"10.5267/j.ijdns.2024.6.012","title":"An improved multi-stage framework for large-scale hierarchical text classification problems using a modified feature hashing and bi-filtering strategy","year":2024,"lang":"en","type":"article","venue":"International Journal of Data and Network Science","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Hierarchy; Reduction (mathematics); Hash function; Pattern recognition (psychology); Artificial intelligence; Dimensionality reduction; Data mining; Feature (linguistics); Multi-label classification; Feature hashing; Task (project management); Scale (ratio); Machine learning; Hash table; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001395618,0.0008728323,0.001748589,0.001921975,0.000762611,0.0009773993,0.002473957,0.001278739,0.002295651],"category_scores_gemma":[0.002490488,0.0004748341,0.001718432,0.002091563,0.0004470241,0.00202337,0.001137154,0.001104541,0.001519262],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000673025,"about_ca_system_score_gemma":0.001534528,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01259562,"about_ca_topic_score_gemma":0.01051042,"domain_scores_codex":[0.9984347,0.0002582663,0.0001422074,0.0003448969,0.0006171872,0.000202727],"domain_scores_gemma":[0.9988425,0.000362147,0.0001056186,0.0001936833,0.000424303,0.00007177122],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002730078,0.0002387872,0.002099408,0.0002042163,0.0001286923,0.0001596618,0.0002034359,0.07527233,0.02970555,0.005967353,0.006052479,0.8796951],"study_design_scores_gemma":[0.00001918666,0.0001119399,0.000883085,0.000007235125,0.00002723997,0.00008532035,0.00003921457,0.9892231,0.004369779,0.003574769,0.001632184,0.00002697811],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.005860639,0.0002339527,0.9924567,0.00006048682,0.00003334333,0.00009414777,0.00008820376,0.0009184629,0.000253977],"genre_scores_gemma":[0.1404907,0.0002529793,0.8540485,0.0001554087,0.000147552,0.0004010396,0.0009478902,0.0001147666,0.003441181],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01259562,"threshold_uncertainty_score":0.02504462,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1216705230253449,"score_gpt":0.3933562205989736,"score_spread":0.2716856975736287,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}