{"id":"W4416679312","doi":"10.48550/arxiv.2504.05058","title":"Not All Data Are Unlearned Equally","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Samsung; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Task (project management); Context (archaeology); Phone; Training set; Data collection","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0190396,0.001775406,0.001711836,0.001572849,0.001389033,0.004474514,0.00246803,0.002966379,0.002402724],"category_scores_gemma":[0.1190059,0.0009238129,0.001700769,0.001748984,0.003229469,0.01490502,0.005251132,0.006922708,0.001402824],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001835325,"about_ca_system_score_gemma":0.001643297,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003039573,"about_ca_topic_score_gemma":0.003975173,"domain_scores_codex":[0.9776295,0.01104929,0.001675705,0.004536539,0.004390032,0.0007189867],"domain_scores_gemma":[0.8845128,0.0722556,0.004393446,0.03232866,0.00503204,0.001477516],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002351468,0.0008445127,0.0688672,0.001474229,0.0009938471,0.0006401901,0.002196145,0.2148169,0.0111377,0.04038132,0.01591364,0.6403828],"study_design_scores_gemma":[0.0001897198,0.0007617564,0.01251323,0.0004971188,0.0003949767,0.001078229,0.001184313,0.6985977,0.03057042,0.2324339,0.02158667,0.0001919707],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2961129,0.003995977,0.6742513,0.006190428,0.0005820304,0.0004726908,0.002758098,0.004920246,0.01071633],"genre_scores_gemma":[0.8769142,0.0007958804,0.1128761,0.001726839,0.0002019865,0.0002431553,0.004479896,0.0006274614,0.002134443],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0190396,"threshold_uncertainty_score":0.1006922,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.255397359614576,"score_gpt":0.3482470129035719,"score_spread":0.0928496532889958,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}