{"meta":{"query_hash":"1c411d0ea448","filters":{"venue":"Theory and Practice of Science and Technology"},"cohort_total":1,"direct_labels_cover":0,"predictions_cover":1,"exported":1,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/1c411d0ea448","api":"https://metacan.xera.ac/api/v1/cohort?venue=Theory+and+Practice+of+Science+and+Technology"},"results":[{"id":"W4412689102","doi":"10.47297/taposatwsp2633-456911.20250605","title":"Optimizing Random Forest with Apache Spark: A Survey on Distributed Machine Learning and Big Data Scalability","year":2025,"lang":"en","type":"article","venue":"Theory and Practice of Science and Technology","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"SPARK (programming language); Scalability; Big data; Random forest; Computer science; Database; Operating system; Data science; Distributed computing; Machine learning; Programming language","score_opus":0.03292571789572647,"score_gpt":0.30944922155147464,"score_spread":0.2765235036557482,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412689102","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013439975,0.06508277,0.90365773,0.0017909774,0.0004677831,0.00015940216,0.00035184534,0.004832126,0.010217361],"genre_scores_gemma":[0.32839817,0.066679314,0.5921652,0.0009739298,0.0015296107,0.00066177663,0.0021991394,0.0027092798,0.0046835714],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99785894,0.0005195967,0.00015290092,0.0003734902,0.0008902134,0.00020484262],"domain_scores_gemma":[0.9980356,0.0009899439,0.00011154699,0.00029181183,0.00048528676,0.00008579989],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034729736,0.001270723,0.002381723,0.0013841176,0.0007644847,0.0016780436,0.0025305864,0.00084157113,0.0016288417],"category_scores_gemma":[0.0061063245,0.00066679064,0.0015240998,0.0037881972,0.00070013874,0.003191518,0.001457145,0.0013247258,0.00083341735],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025590326,0.00021809345,0.003957227,0.0015615978,0.00026606963,0.00017675733,0.00014601817,0.29622293,0.002374611,0.03698489,0.034130648,0.62370527],"study_design_scores_gemma":[0.000090384834,0.00009558384,0.0013620977,0.00017656236,0.000081282495,0.00025204997,0.000111393885,0.8870498,0.0028713383,0.06303135,0.04482242,0.000055781904],"about_ca_topic_score_codex":0.0039877133,"about_ca_topic_score_gemma":0.0029777256,"teacher_disagreement_score":0.0039877133,"about_ca_system_score_codex":0.00084605813,"about_ca_system_score_gemma":0.0019065021,"threshold_uncertainty_score":0.018367052},"labels":[],"label_agreement":null}]}