{"id":"W4415532898","doi":"10.1093/icc/dtaf044","title":"How small is big enough? Open labeled datasets and the development of deep learning","year":2025,"lang":"en","type":"article","venue":"Industrial and Corporate Change","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Institute for Advanced Research","funders":"Canadian Institute for Advanced Research","keywords":"Deep learning; Function (biology); Citation; Object (grammar); Key (lock); Development (topology)","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.04117363,0.0004359054,0.0005590951,0.003221807,0.001687841,0.006539789,0.002028633,0.001765041,0.003214478],"category_scores_gemma":[0.1505771,0.0003556525,0.0003804468,0.004371282,0.005806258,0.01485733,0.003793736,0.003240485,0.000473067],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00362379,"about_ca_system_score_gemma":0.002951399,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004715786,"about_ca_topic_score_gemma":0.006469007,"domain_scores_codex":[0.9829724,0.01148982,0.0005291303,0.00124943,0.00325945,0.0004996441],"domain_scores_gemma":[0.7254523,0.2255175,0.01322496,0.01652908,0.01523404,0.004042219],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001097821,0.0004256943,0.0916172,0.001015553,0.0002461185,0.0002583731,0.002106179,0.03407518,0.002134195,0.5852692,0.05131268,0.2304419],"study_design_scores_gemma":[0.00009362536,0.0001465654,0.02436923,0.001199049,0.0000836766,0.0001482838,0.002552792,0.1196292,0.005368321,0.7868003,0.05951658,0.00009222193],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5081761,0.01800481,0.2484406,0.1563104,0.001890431,0.0002739426,0.01029004,0.0007213521,0.05589228],"genre_scores_gemma":[0.9430887,0.002231177,0.04605122,0.002821409,0.0008354158,0.0001684452,0.002862403,0.0001715497,0.001769674],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9979714,"threshold_uncertainty_score":0.2177495,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2451403693709484,"score_gpt":0.2879666968852078,"score_spread":0.04282632751425935,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}