{"id":"W4417357121","doi":"10.48550/arxiv.2506.10942","title":"Building a Media Ecosystem Observatory from Scratch: Infrastructure, Methodology, and Insights","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Digital ecosystem; Observatory; Search engine indexing; Normalization (sociology); Raw data; Digital media; Social media","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.007520284,0.000913341,0.0006060818,0.004354062,0.002627716,0.006147827,0.002790195,0.0009845641,0.00337965],"category_scores_gemma":[0.02325973,0.0008668631,0.0007351447,0.00479904,0.002298183,0.008144781,0.005945393,0.002335965,0.002566145],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006001632,"about_ca_system_score_gemma":0.02176393,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.2426927,"about_ca_topic_score_gemma":0.2845553,"domain_scores_codex":[0.9952816,0.00115577,0.0003546913,0.001128373,0.001671693,0.000407835],"domain_scores_gemma":[0.9862861,0.002711852,0.0007405551,0.005612136,0.003492953,0.001156383],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000641824,0.001333906,0.09786741,0.001415246,0.0004014425,0.0008124604,0.01124827,0.02833546,0.03875195,0.140732,0.09601571,0.5824443],"study_design_scores_gemma":[0.0002566653,0.0003443027,0.05428911,0.0006202716,0.000193008,0.0004829685,0.008924725,0.4272157,0.06243185,0.1005371,0.3442187,0.00048569],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.05967571,0.0002836123,0.8750973,0.004086713,0.0001613343,0.002087073,0.01597491,0.02973546,0.01289786],"genre_scores_gemma":[0.1463521,0.00029253,0.8253846,0.0003563575,0.00006756433,0.001590517,0.01989128,0.002360214,0.00370483],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9924797,"threshold_uncertainty_score":0.4825602,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1708779774952619,"score_gpt":0.3940047330366261,"score_spread":0.2231267555413643,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}