{"id":"W4224262247","doi":"10.1007/s42488-022-00066-6","title":"Classifying multi-level product categories using dynamic masking and transformer models","year":2022,"lang":"en","type":"article","venue":"Journal of Data Information and Management","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Categorization; Transformer; Generalizability theory; Machine learning; Artificial intelligence; Product (mathematics); Deep learning; Natural language processing; Engineering; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001228939,0.0006886678,0.0006048186,0.002715966,0.000368991,0.001884901,0.001068416,0.0006639966,0.002790456],"category_scores_gemma":[0.002990953,0.000259446,0.001488792,0.001881789,0.0004348035,0.002663397,0.000978585,0.0007664078,0.001272366],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008110413,"about_ca_system_score_gemma":0.0009601059,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006138223,"about_ca_topic_score_gemma":0.005341295,"domain_scores_codex":[0.9992082,0.00013652,0.00006501114,0.0001997647,0.0002599285,0.000130676],"domain_scores_gemma":[0.9984255,0.0006847619,0.0001478307,0.000257559,0.0004100903,0.00007424352],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001734365,0.0005863606,0.03625387,0.0002813386,0.0002537468,0.0003634524,0.000451728,0.1082539,0.02969733,0.03202425,0.004560508,0.7855392],"study_design_scores_gemma":[0.0000130978,0.0001159398,0.004435488,0.00001917817,0.00007490798,0.0001263829,0.000110526,0.9720774,0.004759914,0.01669646,0.001549572,0.00002119882],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2877511,0.0005251887,0.7033918,0.0002694677,0.00009205692,0.0001854338,0.001190208,0.001904168,0.004690638],"genre_scores_gemma":[0.8911952,0.000214171,0.1042057,0.00004943039,0.00002586745,0.00007592599,0.001523196,0.00009898045,0.002611511],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006138223,"threshold_uncertainty_score":0.012205,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1441194875508052,"score_gpt":0.3142199106093222,"score_spread":0.170100423058517,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}