{"id":"W2604773590","doi":"10.71781/11043","title":"Feature selection and term weighting beyond word frequency for calls for tenders documents","year":2006,"lang":"en","type":"dissertation","venue":"Open MIND","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Call for bids; Weighting; Term (time); Feature (linguistics); Word (group theory); Computer science; Selection (genetic algorithm); Feature selection; Information retrieval; Word lists by frequency; Artificial intelligence; tf–idf; Speech recognition; Natural language processing; Mathematics; Linguistics; Business; Procurement; Acoustics; Marketing","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002591225,0.0007713972,0.001058121,0.006962098,0.001009284,0.002700752,0.0009824346,0.001078031,0.003932233],"category_scores_gemma":[0.01438126,0.0003541662,0.001342751,0.007306553,0.0004343284,0.003479084,0.0009697501,0.001296731,0.00291611],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009647893,"about_ca_system_score_gemma":0.001412566,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007133604,"about_ca_topic_score_gemma":0.008931043,"domain_scores_codex":[0.9973333,0.0009012049,0.0003102821,0.0004794683,0.0007387578,0.0002369119],"domain_scores_gemma":[0.9936152,0.00345465,0.0004494998,0.00080669,0.001477402,0.0001965484],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008497476,0.0002806048,0.01003817,0.0004564564,0.0001835697,0.0002353694,0.0003173434,0.006590573,0.02369278,0.006345494,0.02498671,0.9260233],"study_design_scores_gemma":[0.0002782278,0.0007915483,0.1066026,0.0003767007,0.0008535216,0.001855394,0.001445761,0.6554874,0.05625176,0.07467127,0.101009,0.0003768262],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2521948,0.006230497,0.7117627,0.001847339,0.001241416,0.0004851913,0.009664538,0.00590466,0.01066886],"genre_scores_gemma":[0.5743952,0.001575932,0.3892263,0.0001788359,0.000825702,0.0004258302,0.01873416,0.0009141117,0.01372391],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007133604,"threshold_uncertainty_score":0.01418418,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0219019658641877,"score_gpt":0.3165155822977412,"score_spread":0.2946136164335535,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}