{
  "_about": "EPFL MDH research landscape — machine-readable dataset of actors and their public works on misinformation, disinformation and hate speech. Generated from source/content/publications.yaml. Internal outreach/status/working fields are omitted. actors[]: id, name, url, unit, faculty, mdh_focus, dataTypes, techTypes, stage, and publications[].",
  "_source": "https://danakalaaji.github.io/website-epfl-mdh/",
  "_generated": "2026-09-16",
  "_license": "Content for research/educational use; cite the underlying papers via their own DOIs.",
  "actor_count": 71,
  "publication_count": 194,
  "unique_work_count": 152,
  "actors": [
    {
      "id": "west",
      "name": "Robert West",
      "url": "https://people.epfl.ch/robert.west",
      "unit": "DLAB",
      "faculty": "IC",
      "mdh_focus": [
        "M",
        "D",
        "H"
      ],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "LLMs",
        "AI Alignment",
        "NLP",
        "Computational Social Science",
        "Persuasion Analysis",
        "Algorithmic Audit",
        "Causal Inference"
      ],
      "stage": "Prevention + Monitoring + Mitigation",
      "publications": [
        {
          "title": "On the Conversational Persuasiveness of Large Language Models: A Randomized Controlled Trial",
          "wid": "on-the-conversational-persuasiveness-of-large-language-",
          "type": "publication",
          "year": 2025,
          "link": "https://www.nature.com/articles/s41562-025-02194-6",
          "authors": [
            "Francesco Salvi (DLAB, EPFL)",
            "Manoel Horta Ribeiro (DLAB, EPFL)",
            "Riccardo Gallotti (Fondazione Bruno Kessler)",
            "Robert West (DLAB, EPFL)"
          ],
          "epfl_authors": [
            "Francesco Salvi (DLAB, EPFL)",
            "Manoel Horta Ribeiro (DLAB, EPFL)",
            "Robert West (DLAB, EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "LLM persuasion",
            "Microtargeting",
            "GPT-4",
            "Online debates",
            "Influence operations",
            "RCT"
          ],
          "stage": "Monitoring",
          "relevance": 4,
          "about": "Preregistered randomized controlled trial testing whether GPT-4 can out-persuade people in short text debates, and whether giving the model basic sociodemographic facts about its opponent makes it more effective. Participants debated either a human or GPT-4, with or without the opponent's personal data, and opinion change was measured before and after across contentious topics. When GPT-4 had personal information about its opponent, participants were far more likely to shift toward the opposing view than in human-versus-human debates; without personalization, GPT-4 had no significant edge. Personalized GPT-4 raised the odds of a participant shifting toward the opponent's position by 81.2 percent over the human-human baseline.",
          "why": "Provides experimental evidence that personalized LLMs out-persuade humans, a core mechanism of AI-driven disinformation and microtargeting.",
          "data": "Text",
          "themes": [
            "Persuasion & cognitive effects"
          ],
          "subtopics": [
            "Attitude change",
            "LLM persuasion",
            "Microtargeting"
          ],
          "key_terms": [
            "LLM persuasion",
            "Microtargeting",
            "Online debates",
            "Opinion change",
            "Personalized persuasion"
          ],
          "models": [
            "GPT-4"
          ],
          "method_qualifiers": [
            "Randomized controlled trial",
            "Causal inference",
            "Ordinal regression"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "follow_up": [
            {
              "what": "debategpt code repository (epfl-dlab), the analysis and debate-platform code for this study",
              "kind": "repository",
              "url": "https://github.com/epfl-dlab/debategpt",
              "evidence": "Code availability: \"The code to fully reproduce the analyses described in this work is available on GitHub at https://github.com/epfl-dlab/debategpt. Data collection was performed using Empirica v.1.9.5. The study was conducted using Python 3.11, R 4.3.1 and LIWC-22.\" (PMC full text of the Nature Human Behaviour version). The repository page itself states it \"contains code accompanying the research article 'On the Conversational Persuasiveness of GPT-4,' which was published in Nature Human Behaviour.\""
            },
            {
              "what": "debategpt dataset released on Hugging Face: the debate transcripts and pre/post agreement data collected in the trial",
              "kind": "dataset",
              "url": "https://huggingface.co/datasets/frasalvi/debategpt",
              "evidence": "Data availability: \"The debate dataset collected for our study is publicly available at https://huggingface.co/datasets/frasalvi/debategpt\" (PMC full text). The Hugging Face card shows 750 rows in CSV under cc-by-sa-4.0, with pre- and post-treatment agreement fields, the four treatment conditions and ~30 debate topics, matching the paper's design."
            }
          ],
          "data_description": "560 online debates involving 820 US participants.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "United States"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Persuasion & cognitive effects": [
              "LLM persuasion",
              "Microtargeting",
              "Attitude change"
            ]
          },
          "theme_qualifiers_canonical": {
            "Persuasion & cognitive effects": [
              "Attitude change",
              "LLM persuasion",
              "Microtargeting"
            ]
          },
          "tentative": false
        },
        {
          "title": "From Model Training to Model Raising",
          "wid": "from-model-training-to-model-raising-a-call-to-reform-a",
          "type": "publication",
          "year": 2026,
          "venue": "Communications of the ACM 69(2), 24-27",
          "link": "https://arxiv.org/abs/2511.09287",
          "authors": [
            "Roland Aydin (Hamburg Univ. of Technology / Helmholtz-Zentrum Hereon)",
            "Christian Cyron (Hamburg Univ. of Technology / Helmholtz-Zentrum Hereon)",
            "Steve Bachelor (independent)",
            "Ashton Anderson (Univ. of Toronto)",
            "Robert West (EPFL)"
          ],
          "epfl_authors": [
            "Robert West (EPFL)"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_topics": [
            "AI alignment",
            "LLM pre-training",
            "RLHF",
            "Model safety",
            "Value alignment"
          ],
          "stage": "Prevention",
          "relevance": 2,
          "about": "Position paper arguing that the usual recipe for building language models is backwards: a model first absorbs huge amounts of raw internet text from a wild mix of authors, observed without framing, context or deliberate ordering, and only afterward gets alignment added as a corrective layer, a superficial fix applied after deep-rooted cognitive structures have already been formed. As an alternative it proposes model raising, which redesigns the training corpus around four principles: framing data from a first-person perspective, presenting information as lived experience, simulating social interactions so values emerge through exchange, and ordering data so foundational values come early. The aim is a model whose knowledge and values are hard to separate, making it structurally harder to jailbreak into producing harmful output.",
          "why": "It reshapes LLM pre-training so values are embedded early, upstream of harmful-content generation, though it does not study misinformation or hate speech directly.",
          "data": "Text",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Adversarial robustness",
            "Safety alignment",
            "Training-corpus design"
          ],
          "key_terms": [
            "AI alignment",
            "Jailbreaking",
            "Model raising",
            "Intrinsic alignment",
            "AI safety"
          ],
          "models": [],
          "method_qualifiers": [],
          "events_cases": [],
          "built_at_epfl": [],
          "follow_up": [
            {
              "what": "Synthetic Persona Pretraining (SPP), the EPFL dlab system that implements the model-raising proposal by installing an assistant persona from token zero in pretraining, by an overlapping author team (West, Anderson, Aydin)",
              "kind": "successor-work",
              "url": "https://arxiv.org/abs/2608.13482",
              "evidence": "The vision of Model Raising (Aydin et al., 2026) proposes to introduce and shape the assistant from the beginning of training."
            },
            {
              "what": "SPP code release: umbrella repository under the EPFL dlab GitHub organisation, with spp-data, spp-training and spp-evals submodules",
              "kind": "repository",
              "url": "https://github.com/epfl-dlab/spp",
              "evidence": "We publicly release all code, data, and trained models (along with pretraining checkpoints)"
            },
            {
              "what": "SPP model and dataset release: 20 models at 3B and 1.7B (base and instruct) plus 10 datasets on the Hugging Face Hub",
              "kind": "model",
              "url": "https://huggingface.co/dlab-spp",
              "evidence": "Model weights (all five variants, at 3B and 1.7B, base and instruct) and datasets are published on the Hugging Face Hub under dlab-spp."
            }
          ],
          "data_description": "No empirical data (position paper).",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Safety alignment",
              "Training-corpus design",
              "Adversarial robustness"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Adversarial robustness",
              "Safety alignment",
              "Training-corpus design"
            ]
          },
          "tentative": false
        },
        {
          "title": "Can Language Models Recognize Convincing Arguments?",
          "wid": "can-language-models-recognize-convincing-arguments",
          "type": "publication",
          "year": 2024,
          "venue": "Findings of EMNLP 2024",
          "link": "https://aclanthology.org/2024.findings-emnlp.515/",
          "authors": [
            "Paula Dolores Rescala (EPFL)",
            "Manoel Horta Ribeiro (EPFL)",
            "Tiancheng Hu (University of Cambridge)",
            "Robert West (EPFL)"
          ],
          "epfl_authors": [
            "Paula Dolores Rescala (EPFL)",
            "Manoel Horta Ribeiro (EPFL)",
            "Robert West (EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Persuasion",
            "Microtargeting",
            "LLM evaluation",
            "Personalized misinformation",
            "Social sensing",
            "Propaganda"
          ],
          "stage": "Monitoring",
          "relevance": 4,
          "about": "Study of whether LLMs can identify which arguments are persuasive and predict how a particular person will react from their profile and prior beliefs. Several models were tested in zero-shot settings on judging which debater argued better and on predicting a person's stance before and after a debate across contentious topics. On argument quality only GPT-4 matched the human benchmark, at 60.5 percent against 60.7 percent, while the other three models scored between 24.9 and 42.7 percent against a 33.3 percent random baseline; on stance prediction every model performed on a par with crowdworkers. Because the models do well on different debates, stacking their predictions in a supervised logistic regression beats crowdworkers on the two stance tasks, though a plain XGBoost model trained on the same traits still beats every individual model and the stack.",
          "why": "It probes whether LLMs can detect persuasive, demographically tailored arguments, a capability the authors tie to personalized misinformation and propaganda.",
          "data": "Text",
          "themes": [
            "AI safety",
            "Persuasion & cognitive effects"
          ],
          "subtopics": [
            "Capability evaluation",
            "Demographic tailoring",
            "LLM persuasion",
            "Microtargeting",
            "Misuse risk assessment"
          ],
          "key_terms": [
            "LLM persuasion",
            "Political microtargeting",
            "Personalized misinformation",
            "Argument quality",
            "Argument persuasiveness"
          ],
          "models": [
            "GPT-3.5",
            "GPT-4",
            "Llama 2",
            "Mistral 7B"
          ],
          "method_qualifiers": [
            "Zero-shot prompting",
            "Model stacking"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "PoliIssues",
              "kind": "dataset",
              "url": "https://zenodo.org/records/13887286",
              "evidence": "Paper: \"Hereafter, we call this dataset the PoliIssues dataset.\" Repository README (reached from the paper's own link go.epfl.ch/persuasion-llm): \"Data and Code for: 'Can Language Models Recognize Convincing Arguments?'\" and \"Data for this work is available through Zenodo (https://zenodo.org/records/13887286).\" The Zenodo record is titled \"Can Language Models Recognize Convincing Arguments?\" by Paula Rescala and Manoel Horta Ribeiro (Ecole Polytechnique Federale de Lausanne)."
            },
            {
              "name": "PoliProp",
              "kind": "dataset",
              "url": "https://zenodo.org/records/13887286",
              "evidence": "Paper: \"Hereafter, we call this the PoliProp dataset.\" Repository README (reached from the paper's own link go.epfl.ch/persuasion-llm): \"Data and Code for: 'Can Language Models Recognize Convincing Arguments?'\" and \"Data for this work is available through Zenodo (https://zenodo.org/records/13887286).\" The Zenodo record is titled \"Can Language Models Recognize Convincing Arguments?\" by Paula Rescala and Manoel Horta Ribeiro (Ecole Polytechnique Federale de Lausanne)."
            }
          ],
          "follow_up": [
            {
              "what": "debate-gpt-x, the public data-and-code repository released with the paper (analysis notebook, the debate_gpt package, prompt/LLM-output pipeline)",
              "kind": "repository",
              "url": "https://github.com/manoelhortaribeiro/debate-gpt-x",
              "evidence": "README first line: \"# Data and Code for: 'Can Language Models Recognize Convincing Arguments?'\" and \"In order to reproduce our results, only the data in `tidy.zip` on Zenodo is needed... All the analysis is done directly in the notebook `analyses.ipynb`.\""
            }
          ],
          "data_description": "833 annotated debate.org political debates, 4871 votes, 751 crowdsourced labels.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "debate.org"
          ],
          "region_country": [
            "United States"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Misuse risk assessment",
              "Capability evaluation"
            ],
            "Persuasion & cognitive effects": [
              "Microtargeting",
              "LLM persuasion",
              "Demographic tailoring"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Misuse risk assessment"
            ],
            "Persuasion & cognitive effects": [
              "LLM persuasion",
              "Microtargeting"
            ]
          },
          "tentative": false
        },
        {
          "title": "A Logical Fallacy-Informed Framework for Argument Generation",
          "wid": "a-logical-fallacy-informed-framework-for-argument-gener",
          "type": "publication",
          "year": 2025,
          "venue": "NAACL 2025",
          "link": "https://aclanthology.org/2025.naacl-long.374/",
          "authors": [
            "Luca Mouchel",
            "Debjit Paul",
            "Shaobo Cui",
            "Robert West",
            "Antoine Bosselut",
            "Boi Faltings"
          ],
          "epfl_authors": [
            "Luca Mouchel",
            "Debjit Paul",
            "Shaobo Cui",
            "Robert West",
            "Antoine Bosselut",
            "Boi Faltings"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "Logical fallacies",
            "Argument generation",
            "LLM alignment",
            "Preference optimization",
            "Misinformation prevention"
          ],
          "stage": "Prevention",
          "relevance": 2,
          "about": "Study treating the tendency of LLMs to produce arguments with logical fallacies as a misinformation risk, since flawed reasoning is harder for readers to catch than an outright false statement. It proposes a training method, FIPO, that adds a fallacy classification objective across 13 fallacy categories and weights the penalty by how common each fallacy is in real data, lowering fallacy rates to 17 percent for Llama-2 and 19.5 percent for Mistral and largely fixing the most stubborn category, faulty generalization. The authors flag the framework as dual use, since the same approach could generate more coherent but deliberately deceptive arguments.",
          "why": "It reduces fallacious LLM-generated arguments the authors link to spreading misinformation, while flagging a dual-use risk.",
          "data": "Text",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Fallacy-aware alignment"
          ],
          "key_terms": [
            "Logical fallacies",
            "Argument generation",
            "LLM alignment",
            "Preference optimization",
            "AI-generated misinformation risk"
          ],
          "models": [
            "ChatGPT",
            "ELECTRA",
            "GPT-4",
            "Llama-2",
            "Mistral"
          ],
          "method_qualifiers": [
            "LoRA",
            "LLM-as-a-judge",
            "Preference optimisation",
            "DPO"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Fallacy-augmented preference dataset",
              "kind": "dataset"
            },
            {
              "name": "FIPO (Fallacy-Informed Preference Optimization)",
              "kind": "framework"
            },
            {
              "name": "Fallacy preference dataset",
              "kind": "dataset",
              "url": "https://github.com/lucamouchel/Logical-Fallacies/tree/main/data",
              "evidence": "Paper footnote 1, page 1: \"Our code and datasets are publicly available for research purposes at github.com/lucamouchel/Logical-Fallacies\". The repository's data/ directory (opened) contains the folders argumentation/, generated/, preference-data/, sft/, sft_rag/ and the file test_debate.txt, matching the paper's pipeline (ExplaGraphs argumentation data plus ChatGPT-generated fallacious counterparts forming the preference pairs)."
            },
            {
              "name": "FIPO",
              "kind": "framework",
              "url": "https://github.com/lucamouchel/Logical-Fallacies",
              "evidence": "GitHub repository description reads \"A Logical Fallacy-Informed Framework for Argument Generation\"; the README credits Luca Mouchel, Debjit Paul, Shaobo Cui, Robert West, Antoine Bosselut and Boi Faltings, NAACL 2025, and the src/ tree carries the data collection, supervised fine-tuning, preference optimization (DPO, KTO, PPO, CPO) and FIPO stages described in the paper."
            }
          ],
          "follow_up": [
            {
              "what": "Official code and data release for the paper (GitHub: lucamouchel/Logical-Fallacies)",
              "kind": "repository",
              "url": "https://github.com/lucamouchel/Logical-Fallacies",
              "evidence": "Paper footnote 1: \"Our code and datasets are publicly available for research purposes at github.com/lucamouchel/Logical-Fallacies\" - confirmed by opening the repository, whose description is the paper title and which implements the FIPO training and evaluation pipeline."
            }
          ],
          "data_description": "7872 generated fallacious arguments over ExplaGraphs topics and stances.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "United States"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Fallacy-aware alignment"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Fallacy-aware alignment"
            ]
          },
          "tentative": false
        },
        {
          "title": "Auditing Radicalization Pathways on YouTube",
          "wid": "auditing-radicalization-pathways-on-youtube",
          "type": "publication",
          "year": 2019,
          "link": "https://arxiv.org/abs/1908.08313",
          "authors": [
            "Manoel Horta Ribeiro (EPFL)",
            "Raphael Ottoni (UFMG)",
            "Robert West (EPFL)",
            "Virgílio A. F. Almeida (UFMG)",
            "Wagner Meira Jr. (UFMG)"
          ],
          "epfl_authors": [
            "Manoel Horta Ribeiro (EPFL)",
            "Robert West (EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "H"
          ],
          "mdh_topics": [
            "Radicalization",
            "YouTube",
            "Recommender systems",
            "Hate speech",
            "Alt-right",
            "Algorithmic audit"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "First large-scale quantitative test of the claim that YouTube acts as a radicalization pipeline moving viewers from mild contrarian content toward far-right extremism. Using videos, channels, and millions of comments sorted into mainstream media, the Intellectual Dark Web, the Alt-lite, and the white-supremacist Alt-right, it traces whether people drift from milder to more extreme content. About 12 percent of users who first commented only on Alt-lite or Intellectual Dark Web content had moved to Alt-right content within a year, three to four times the rate for users who started on mainstream media. Channel recommendations made more extreme content easy to reach from milder communities.",
          "why": "First large-scale audit linking YouTube recommendations to user migration into white-supremacist content, squarely in hate speech and radicalization.",
          "data": "Text, channel networks",
          "themes": [
            "Radicalisation & violent extremism",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Cross-channel activity",
            "Echo chambers",
            "Far-right communities",
            "Radicalisation pathways",
            "Recommender amplification",
            "User migration"
          ],
          "key_terms": [
            "Alt-right",
            "Intellectual Dark Web",
            "Radicalization pipeline",
            "User migration",
            "Algorithmic auditing"
          ],
          "models": [],
          "method_qualifiers": [
            "Algorithmic auditing",
            "Random walk simulation",
            "Manual annotation",
            "Statistical & causal analysis"
          ],
          "events_cases": [
            "2016 US presidential election",
            "Alt-right movement",
            "YouTube radicalization"
          ],
          "built_at_epfl": [],
          "follow_up": [
            {
              "what": "Analysis code released as the GitHub repository manoelhortaribeiro/radicalization_youtube (four Jupyter notebooks reproducing the paper's figures and tables); the underlying data is withheld and shared on request",
              "kind": "repository",
              "url": "https://github.com/manoelhortaribeiro/radicalization_youtube",
              "evidence": "Code for the paper \"Auditing Radicalization Pathways on YouTube\" (FAT* 2020) ... Due to the sensitivity of the data, the data/helpers necessary to reproduce the analyses are not made available here. We are, however, willing to consider sharing it with other research groups upon request :)"
            },
            {
              "what": "The YouTube dataset was extended by the same group in \"Are Anti-Feminist Communities Gateways to the Far Right? Evidence from Reddit and YouTube\" (Mamie, Horta Ribeiro, West, WebSci 2021), which reuses the Alt-right / Alt-lite / I.D.W. channel data and collection methodology to test whether the Manosphere feeds the far right",
              "kind": "successor-work",
              "url": "https://arxiv.org/abs/2102.12837",
              "evidence": "To study YouTube, we expand the dataset obtained from (Ribeiro et al. 2020). We leverage the same methodology to collect data associated with 4 groups in the Manosphere ... Notice that we use this data along with the YouTube data published by Ribeiro et al. (Ribeiro et al. 2020), which was captured in a similar fashion."
            }
          ],
          "data_description": "330k videos, 349 channels, 72M comments, 2M recommendations.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "YouTube"
          ],
          "region_country": [
            "United States"
          ],
          "targeted_group": [
            "Ethnic minorities",
            "Religious minorities",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Radicalisation & violent extremism": [
              "Far-right communities",
              "Radicalisation pathways",
              "User migration"
            ],
            "Spread, amplification & networks": [
              "Recommender amplification",
              "Echo chambers",
              "Cross-channel activity"
            ]
          },
          "theme_qualifiers_canonical": {
            "Radicalisation & violent extremism": [
              "Far-right communities",
              "Radicalisation pathways",
              "User migration"
            ],
            "Spread, amplification & networks": [
              "Cross-platform migration",
              "Echo chambers",
              "Recommender amplification"
            ]
          },
          "tentative": false
        },
        {
          "title": "Message Distortion in Information Cascades",
          "wid": "message-distortion-in-information-cascades",
          "type": "publication",
          "year": 2019,
          "link": "https://arxiv.org/abs/1902.09197",
          "authors": [
            "Manoel Horta Ribeiro (UFMG)",
            "Kristina Gligorić (EPFL)",
            "Robert West (EPFL)"
          ],
          "epfl_authors": [
            "Kristina Gligorić (EPFL)",
            "Robert West (EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M"
          ],
          "mdh_topics": [
            "Information cascades",
            "Misinformation",
            "Science communication",
            "Summarisation",
            "Telephone effect"
          ],
          "stage": "Monitoring",
          "relevance": 4,
          "about": "Crowdsourced experiment on how information gets warped as it passes from person to person even when nobody is lying, a telephone effect where small errors accumulate over successive retellings. Workers iteratively summarized medical abstracts, either using the previous person's summary as input or always starting from the original. Cascading distorts the most important part the most: peripheral details survive, but an abstract's core conclusion is represented about 25 percentage points less often than in direct compression, and the usual advantage of domain expertise disappears.",
          "why": "It shows experimentally how accurate information becomes misinformation through cascading retelling alone, without any malicious actor.",
          "data": "Text",
          "themes": [
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Information cascades",
            "Iterative summarization",
            "Message distortion"
          ],
          "key_terms": [
            "Information cascades",
            "Message distortion",
            "Science communication",
            "Telephone effect",
            "Medical misinformation"
          ],
          "models": [],
          "method_qualifiers": [
            "Controlled experiment",
            "Crowdsourced annotation",
            "Statistical & causal analysis"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Message distortion in information cascades dataset (mdic)",
              "kind": "dataset"
            }
          ],
          "follow_up": [
            {
              "what": "epfl-dlab/mdic, the code and data repository for the paper",
              "kind": "repository",
              "url": "https://github.com/epfl-dlab/mdic",
              "evidence": "Repo README: 'This repository contains the data and code of the paper \"Message Distortion in Information Cascades\"', followed by the BibTeX entry naming Ribeiro, Gligoric and West, Proceedings of the 2019 World Wide Web Conference. The paper itself names the repo in a first-page footnote: 'Code/data: github.com/epfl-dlab/mdic'."
            },
            {
              "what": "Interactive cascade-visualisation website released alongside the data",
              "kind": "deployment",
              "url": "https://epfl-dlab.github.io/mdic/",
              "evidence": "Repo README: 'Check out the accompanying website which allows you to visualize the data.' linking to https://epfl-dlab.github.io/mdic/. The live page is titled 'Message Distortion in Information Cascades' and lists the study's medical abstracts (Breast Cancer, Immunization, ...)."
            },
            {
              "what": "The paper's MTurk iterative-summarization task was reused by the same lab to measure LLM use by crowd workers: 'Artificial Artificial Artificial Intelligence: Crowd Workers Widely Use Large Language Models for Text Production Tasks' (Veselovsky, Horta Ribeiro, West), later published in CACM 2024 as 'Prevalence and Prevention of Large Language Model Use in Crowd Work'",
              "kind": "successor-work",
              "url": "https://arxiv.org/abs/2306.07899",
              "evidence": "'We modify a prior MTurk task originally devised by Horta Ribeiro et al. 2019, whose goal was to study the so-called \"telephone effect,\" whereby information is gradually lost or distorted as a message is passed from human to human in an information cascade.' and 'In the original study, crowd workers produced eight increasingly short summaries of each original abstract, forming entire information cascades. For our purpose, however, we reduced the task to a single summarization step'."
            }
          ],
          "data_description": "Crowdsourced summaries of 16 NEJM abstracts across 5 target lengths.",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "Spread, amplification & networks": [
              "Message distortion",
              "Information cascades",
              "Iterative summarization"
            ]
          },
          "theme_qualifiers_canonical": {
            "Spread, amplification & networks": [
              "Information cascades",
              "Message distortion"
            ]
          },
          "tentative": false
        },
        {
          "title": "Analyzing the \"Sleeping Giants\" Activism Model in Brazil",
          "wid": "analyzing-the-sleeping-giants-activism-model-in-brazil",
          "type": "publication",
          "year": 2022,
          "venue": "14th ACM Web Science Conference (WebSci 2022), 87-97",
          "link": "https://arxiv.org/abs/2105.07523",
          "authors": [
            "Bárbara Gomes Ribeiro (UFMG)",
            "Manoel Horta Ribeiro (EPFL)",
            "Virgílio Almeida (Harvard University)",
            "Wagner Meira Jr. (UFMG)"
          ],
          "epfl_authors": [
            "Manoel Horta Ribeiro (EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M"
          ],
          "mdh_topics": [
            "Demonetisation",
            "Online activism",
            "Ad-revenue",
            "Twitter",
            "Brazil",
            "Counter-misinformation"
          ],
          "stage": "Mitigation",
          "relevance": 5,
          "about": "Study of the Sleeping Giants Brasil playbook, an online activism model that pressures companies to pull advertising from outlets spreading fake news and hate speech. Analyzing three campaigns with the group's tweets and the targeted companies using causal-inference methods, it finds the boycott requests worked at the company level: initial requests succeeded in 83.85 percent of cases, with most responses coming within a week. The campaigns produced no significant change in the targeted outlets' audience engagement or search interest over the following six months. Ad-revenue starvation succeeds in getting advertisers to leave, while reducing the outlets' actual reach does not.",
          "why": "It measures the real-world effectiveness of an advertising-boycott campaign aimed at outlets spreading fake news and hate speech.",
          "data": "Text (Portuguese Twitter + Google Trends); 1560 tweets from @slpng_giants_pt, 192 targeted companies, 3 SGB campaigns May-Sept 2020",
          "themes": [
            "Content moderation & enforcement",
            "Economics & incentives of MDH"
          ],
          "subtopics": [
            "Ad-revenue disruption",
            "Advertiser boycotts",
            "Consumer-led enforcement",
            "Demonetisation"
          ],
          "key_terms": [
            "Sleeping Giants",
            "Online activism",
            "Advertiser boycott",
            "Brazil",
            "Advertising boycott"
          ],
          "models": [
            "Perspective API",
            "SentiStrength"
          ],
          "method_qualifiers": [
            "Causal inference",
            "Sentiment analysis",
            "Synthetic control",
            "Toxicity classification"
          ],
          "events_cases": [
            "2020 Brazilian political climate",
            "COVID-19 pandemic",
            "Sleeping Giants Brasil campaigns (2020)"
          ],
          "built_at_epfl": [],
          "data_description": "1560 SGB tweets, 166 companies, about 1.1M mention tweets, 2020.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "region_country": [
            "Brazil"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Demonetisation",
              "Consumer-led enforcement"
            ],
            "Economics & incentives of MDH": [
              "Ad-revenue disruption",
              "Advertiser boycotts"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Demonetisation"
            ],
            "Economics & incentives of MDH": [
              "Ad-revenue disruption"
            ]
          },
          "tentative": false
        },
        {
          "title": "Deplatforming Norm-Violating Influencers on Social Media Reduces Overall Online Attention Toward Them",
          "wid": "deplatforming-norm-violating-influencers-on-social-medi",
          "type": "publication",
          "year": 2025,
          "venue": "CSCW 2025",
          "link": "https://arxiv.org/abs/2401.01253",
          "authors": [
            "Manoel Horta Ribeiro (EPFL)",
            "Shagun Jhaver (Rutgers)",
            "Jordi Cluet i Martinell (EPFL)",
            "Marie Reignier-Tayar (EPFL)",
            "Robert West (EPFL)"
          ],
          "epfl_authors": [
            "Manoel Horta Ribeiro (EPFL)",
            "Jordi Cluet i Martinell (EPFL)",
            "Marie Reignier-Tayar (EPFL)",
            "Robert West (EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_topics": [
            "Deplatforming",
            "Content moderation",
            "Influencers",
            "Misinformation",
            "Online attention",
            "Causal inference"
          ],
          "stage": "Mitigation",
          "relevance": 4,
          "about": "Quasi-experimental study of what happens to a norm-violating influencer's overall online presence after removal from a major platform, addressing the worry that bans simply push attention elsewhere. Using a large dataset of deplatforming events, it tracks platform-agnostic attention through Google search interest and Wikipedia pageviews. Twelve months after deplatforming, attention had fallen by 63 percent on Google and 43 percent on Wikipedia, with people banned specifically for spreading misinformation reduced further than those banned for other reasons. Both permanent and temporary bans were effective.",
          "why": "It provides causal evidence that deplatforming reduces attention to norm-violating influencers, including those banned for misinformation.",
          "data": "Text/Web metrics; 165 deplatforming events / 101 influencers; Google Trends, Wikipedia pageviews, Media Cloud",
          "themes": [
            "Content moderation & enforcement"
          ],
          "subtopics": [
            "Ban effectiveness",
            "Deplatforming",
            "Temporary bans"
          ],
          "key_terms": [
            "Deplatforming",
            "Online attention",
            "Influencers",
            "Content moderation",
            "Difference-in-differences"
          ],
          "models": [],
          "method_qualifiers": [
            "Causal inference",
            "Difference-in-differences"
          ],
          "events_cases": [
            "Alex Jones ban"
          ],
          "built_at_epfl": [
            {
              "name": "Deplatforming events and online-attention dataset",
              "kind": "dataset"
            },
            {
              "name": "Deplatforming events dataset",
              "kind": "dataset",
              "url": "https://github.com/epfl-dlab/deplatforming_influencers",
              "evidence": "README of github.com/epfl-dlab/deplatforming_influencers: \"# Deplatforming Influencers / Code and Data for \\\"Deplatforming Norm-Violating Influencers on Social Media Reduces Overall Online Attention Toward Them.\\\"\" and, under **Data:**, \"1. `banned_unfiltered.csv`: Dataset of banned users (unfiltered). 2. `banned_final.csv`: Dataset of banned users (filtered). 3. `joined_final.csv`: Attention traces of banned users (linked to #2 above).\""
            }
          ],
          "data_description": "165 deplatforming events, 101 influencers, attention traces 2016-2021.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Facebook",
            "Instagram",
            "Reddit",
            "Twitter/X",
            "YouTube"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Deplatforming",
              "Ban effectiveness",
              "Temporary bans"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Ban effectiveness",
              "Deplatforming"
            ]
          },
          "tentative": false
        },
        {
          "title": "Can online attention signals help fact-checkers fact-check?",
          "wid": "can-online-attention-signals-help-fact-checkers-fact-ch",
          "type": "publication",
          "year": 2022,
          "venue": "MEDIATE workshop at ICWSM 2022",
          "link": "https://arxiv.org/abs/2109.09322",
          "authors": [
            "Manoel Horta Ribeiro",
            "Savvas Zannettou",
            "Oana Goga",
            "Fabrício Benevenuto",
            "Robert West"
          ],
          "epfl_authors": [
            "Manoel Horta Ribeiro",
            "Robert West"
          ],
          "about": "Paper proposing a framework to study fact-checking with online attention signals, which extracts claims from fact-checks, links them with knowledge graph entities and estimates the attention these entities receive, using Google Trends. A preliminary study applying the framework to 879 COVID-19-related fact-checks done in 2020 by 81 international organizations suggests that there is often a disconnect between attention and fact-checking: in around 40 percent of countries that fact-checked ten or more claims, half or more of the ten most popular claims were not fact-checked. Claims were first fact-checked after receiving, on average, 35 percent of the total online attention they would eventually receive in 2020, with considerable variation among claims.",
          "themes": [
            "Spread, amplification & networks",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Check-worthiness ranking",
            "Cross-country claim migration",
            "Exposure intensity",
            "Information cascades"
          ],
          "mdh_focus": [
            "M"
          ],
          "mdh_relevance": "direct",
          "relevance": 5,
          "stage": "Monitoring + Mitigation",
          "key_terms": [
            "COVID-19 misinformation",
            "Google Trends",
            "Fact-checking",
            "Online attention",
            "Attention life cycle"
          ],
          "models": [],
          "method_qualifiers": [
            "Entity linking",
            "DBSCAN clustering",
            "Time-series analysis"
          ],
          "events_cases": [
            "COVID-19 pandemic"
          ],
          "built_at_epfl": [
            {
              "name": "Framework for studying fact-checking with online attention signals",
              "kind": "framework"
            }
          ],
          "data_description": "879 COVID-19 fact-checks across 72 countries, 2586 Google Trends time series.",
          "platform": [
            "Google Search"
          ],
          "region_country": [
            "Global",
            "International"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Spread, amplification & networks": [
              "Exposure intensity",
              "Information cascades",
              "Cross-country claim migration"
            ],
            "Verification & content authenticity": [
              "Check-worthiness ranking"
            ]
          },
          "theme_qualifiers_canonical": {
            "Spread, amplification & networks": [
              "Exposure intensity",
              "Information cascades"
            ],
            "Verification & content authenticity": [
              "Check-worthiness ranking"
            ]
          },
          "tentative": false
        },
        {
          "title": "A Glitch in the Matrix? Locating and Detecting Language Model Grounding with Fakepedia",
          "wid": "a-glitch-in-the-matrix-locating-and-detecting-language-",
          "type": "publication",
          "year": 2024,
          "venue": "ACL 2024 (Long Papers)",
          "link": "https://doi.org/10.18653/v1/2024.acl-long.369",
          "authors": [
            "Giovanni Monea",
            "Maxime Peyrard",
            "Martin Josifoski",
            "Vishrav Chaudhary",
            "Jason Eisner",
            "Emre Kıcıman",
            "Hamid Palangi",
            "Barun Patra",
            "Robert West"
          ],
          "epfl_authors": [
            "Giovanni Monea",
            "Martin Josifoski",
            "Robert West"
          ],
          "about": "Study of how large language models ground answers in context that contradicts factual knowledge stored in their parameters, using Fakepedia, a new dataset of counterfactual texts constructed to clash with a model's internal knowledge. The authors benchmark various LLMs on it and propose Masked Grouped Causal Tracing (MGCT), a causal mediation analysis of model components. Applied to Llama2-7B and GPT2-XL, MGCT indicates that grounding, contrary to factual recall, may be a distributed process, and that MLPs on the last subject token have a high effect when answers are ungrounded. An XGBoost classifier built on MGCT features of GPT2-XL responses distinguishes grounded from ungrounded responses with 92.8 percent test accuracy.",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Factual robustness",
            "Interpretability and oversight",
            "Misuse risk assessment"
          ],
          "mdh_focus": [
            "M"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "Monitoring",
          "key_terms": [
            "Retrieval-augmented generation",
            "Mechanistic interpretability",
            "LLM grounding",
            "Factual recall",
            "Causal mediation analysis"
          ],
          "models": [
            "GPT-2 XL",
            "GPT-3.5 Turbo",
            "GPT2-XL",
            "LLaMA-7B",
            "Llama2-13B",
            "Mistral-7B",
            "Zephyr-7B"
          ],
          "method_qualifiers": [
            "Causal tracing",
            "Explainability & interpretability"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Fakepedia",
              "kind": "dataset"
            },
            {
              "name": "Masked Grouped Causal Tracing (MGCT)",
              "kind": "tool"
            }
          ],
          "data_description": "6090 Fakepedia-base paragraphs and 5340 multi-hop, plus LLM responses.",
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Factual robustness",
              "Interpretability and oversight",
              "Misuse risk assessment"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Factual robustness",
              "Interpretability and oversight",
              "Misuse risk assessment"
            ]
          },
          "tentative": false
        },
        {
          "title": "Automated Content Moderation Increases Adherence to Community Guidelines",
          "wid": "automated-content-moderation-increases-adherence-to-com",
          "type": "publication",
          "year": 2023,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2210.10454",
          "authors": [
            "Manoel Horta Ribeiro",
            "Justin Cheng",
            "Robert West"
          ],
          "epfl_authors": [
            "Manoel Horta Ribeiro",
            "Robert West"
          ],
          "about": "Study of 412 million Facebook comments measuring how automated content moderation, enforcing community guidelines for violence and incitement, affects subsequent rule-breaking behavior and engagement. Using public comments by adult U.S. users from June to August 2022, it applies a fuzzy regression discontinuity design around the classifier score thresholds above which comments are hidden or deleted. Deleting comments decreased subsequent rule-breaking in threads with 20 or fewer comments, even among other participants, and its effect on affected users' rule-breaking outlasted its effect on their commenting, while hiding content had small and statistically insignificant effects.",
          "themes": [
            "Content moderation & enforcement"
          ],
          "subtopics": [
            "Automated pre-filtering",
            "Ban effectiveness",
            "Deplatforming"
          ],
          "mdh_focus": [
            "H"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "Mitigation",
          "key_terms": [
            "Automated content moderation",
            "Violence and incitement",
            "Rule-breaking behavior",
            "Community guidelines",
            "Anti-social behavior"
          ],
          "models": [],
          "method_qualifiers": [
            "Causal inference",
            "Fuzzy regression discontinuity"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "412M Facebook comments, 1.5M posts, 1.3M users, US, 2022.",
          "platform": [
            "Facebook"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Ban effectiveness",
              "Automated pre-filtering",
              "Deplatforming"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Automated pre-filtering",
              "Ban effectiveness",
              "Deplatforming"
            ]
          },
          "tentative": false
        },
        {
          "title": "Linguistic effects on news headline success: Evidence from thousands of online field experiments",
          "wid": "linguistic-effects-on-news-headline-success-evidence-fr",
          "type": "publication",
          "year": 2023,
          "venue": "PLOS ONE (Registered Report)",
          "link": "https://doi.org/10.1371/journal.pone.0281682",
          "authors": [
            "Kristina Gligorić",
            "George Lifchits",
            "Robert West",
            "Ashton Anderson"
          ],
          "epfl_authors": [
            "Kristina Gligorić",
            "Robert West"
          ],
          "about": "Registered report on the linguistic features of news headline success, using field experiments (A/B tests) on Upworthy.com comparing headline variants for the same articles. Hypotheses based on prior literature and 5048 pilot headline pairs were tested on 24333 held-out pairs, where linguistic features predicted the more successful headline with 54.42 percent accuracy (random guessing: 50 percent). Negative emotion words, length, indefinite articles, first-person singular and third-person pronouns were positively associated with success and first-person plural pronouns negatively, with no evidence for the hypothesized effects of positive emotion words, readability, the definite article or second-person pronouns. The authors note that the very format of a short headline might favour negativity.",
          "themes": [
            "Persuasion & cognitive effects"
          ],
          "subtopics": [
            "Clickbait design",
            "Framing and emotional appeals",
            "Negativity bias"
          ],
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "NA",
          "key_terms": [
            "News headline success",
            "Clickbait",
            "A/B testing",
            "Linguistic features",
            "Click-through rate"
          ],
          "models": [],
          "method_qualifiers": [
            "Causal inference",
            "Logistic regression",
            "A/B testing"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "24333 comparable headline pairs from 32487 Upworthy A/B tests, 2013–2015.",
          "platform": [
            "Upworthy"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Persuasion & cognitive effects": [
              "Framing and emotional appeals",
              "Negativity bias",
              "Clickbait design"
            ]
          },
          "theme_qualifiers_canonical": {
            "Persuasion & cognitive effects": [
              "Framing and emotional appeals"
            ]
          },
          "tentative": false
        },
        {
          "title": "Post Guidance for Online Communities",
          "wid": "post-guidance-for-online-communities",
          "type": "publication",
          "year": 2025,
          "venue": "CSCW 2025",
          "link": "https://arxiv.org/abs/2411.16814",
          "authors": [
            "Manoel Horta Ribeiro",
            "Robert West",
            "Ryan Lewis",
            "Sanjay Kairam"
          ],
          "epfl_authors": [
            "Manoel Horta Ribeiro",
            "Robert West"
          ],
          "about": "Randomized experiment evaluating post guidance, a moderation approach where community rules trigger interventions, such as showing a message or preventing submission, while users draft a post. Tested on Reddit with 97616 posters in 33 subreddits over 63 days, the feature increased non-removed posts by 5.8 percent, cut reports by 9.4 percent and AutoModerator removals by 34.9 percent, and raised the comments, screen views and upvotes posts received, even though fewer posts were started and submitted. It did not increase user participation, worked similarly for newcomers and veterans, and brought the biggest increases in non-removed posts to communities that set up many rules or relied heavily on AutoModerator.",
          "themes": [
            "Content moderation & enforcement"
          ],
          "subtopics": [
            "Automated pre-filtering",
            "Community-specific rules",
            "Proactive moderation"
          ],
          "mdh_focus": [
            "M",
            "H"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "Prevention",
          "key_terms": [
            "Proactive content moderation",
            "Post Guidance",
            "Moderator workload",
            "Content moderation",
            "Reddit"
          ],
          "models": [],
          "method_qualifiers": [
            "Randomized controlled trial",
            "Causal inference",
            "Field experiment",
            "Poisson regression"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Post Guidance",
              "kind": "tool"
            }
          ],
          "data_description": "97616 users, 33 subreddits, 63-day randomized field experiment.",
          "platform": [
            "Reddit"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Proactive moderation",
              "Automated pre-filtering",
              "Community-specific rules"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Automated pre-filtering"
            ]
          },
          "tentative": false
        },
        {
          "title": "Protection from Evil and Good: The Differential Effects of Page Protection on Wikipedia Article Quality",
          "wid": "protection-from-evil-and-good-the-differential-effects-",
          "type": "publication",
          "year": 2025,
          "venue": "ICWSM 2025",
          "link": "https://doi.org/10.1609/icwsm.v19i1.35896",
          "authors": [
            "Thorsten Ruprechter",
            "Manoel Horta Ribeiro",
            "Robert West",
            "Denis Helic"
          ],
          "epfl_authors": [
            "Robert West"
          ],
          "about": "Quasi-experimental study of how page protection, which restricts who can edit an article, affects article quality on the English Wikipedia. Using decade-long data, it matches articles protected after a request for page protection with similar articles whose request was declined, and applies a difference-in-differences approach to an automated quality metric. The effect depends on the characteristics of the article before the intervention: high-quality articles are affected positively and low-quality articles negatively, and subsequent analysis suggests high-quality articles degrade when left unprotected whereas low-quality articles improve. The effect also varies across topics, with no notable effect on STEM articles.",
          "themes": [
            "Content moderation & enforcement"
          ],
          "subtopics": [
            "Ban effectiveness",
            "Moderation effect heterogeneity",
            "Page protection"
          ],
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "Mitigation",
          "key_terms": [
            "Wikipedia page protection",
            "Article quality",
            "Vandalism",
            "Page protection",
            "Content moderation"
          ],
          "models": [
            "ORES",
            "ORES articlequality",
            "ORES articletopic"
          ],
          "method_qualifiers": [
            "Difference-in-differences",
            "Propensity score matching",
            "Causal inference"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "English Wikipedia page protection and RfPP data with ORES scores",
              "kind": "dataset"
            }
          ],
          "data_description": "299k page protections, 127k requests, English Wikipedia 2012-2023",
          "platform": [
            "Wikipedia"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Page protection",
              "Moderation effect heterogeneity",
              "Ban effectiveness"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Ban effectiveness"
            ]
          },
          "tentative": false
        },
        {
          "title": "Stranger Danger! Cross-Community Interactions with Fringe Users Increase the Growth of Fringe Communities on Reddit",
          "wid": "stranger-danger-cross-community-interactions-with-fring",
          "type": "publication",
          "year": 2023,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2310.12186",
          "authors": [
            "Giuseppe Russo",
            "Manoel Horta Ribeiro",
            "Robert West"
          ],
          "epfl_authors": [
            "Manoel Horta Ribeiro",
            "Robert West"
          ],
          "about": "Study applying text-based causal inference to test whether fringe-interactions, comment exchanges between members and non-members of fringe communities, draw new members to r/Incels, r/GenderCritical and r/The_Donald on Reddit. Users who received such interactions were up to 4.2 percentage points more likely to join than similar matched users, and interactions using toxic language had a 5 percentage point higher chance of attracting newcomers than non-toxic ones. The effect varied with the communities where interactions happened, such as left or right-leaning ones; repeated for non-fringe communities, effects were smaller and not statistically significant. An estimated 7.2, 3.1 and 2.3 percent of newcomers to the three communities joined after such interactions.",
          "themes": [
            "Radicalisation & violent extremism",
            "Spread, amplification & networks",
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Comment toxicity scoring",
            "Community infiltration",
            "Radicalisation pathways",
            "User migration"
          ],
          "mdh_focus": [
            "H"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "Monitoring",
          "key_terms": [
            "Fringe communities",
            "Incels",
            "Reddit",
            "Alt-right",
            "Capitol riot"
          ],
          "models": [
            "BERT",
            "Perspective API"
          ],
          "method_qualifiers": [
            "Causal inference",
            "Propensity score matching"
          ],
          "events_cases": [
            "r/GenderCritical",
            "r/Incels",
            "r/The Donald"
          ],
          "built_at_epfl": [],
          "data_description": "Reddit comments and posts from ~15M fringe and 5M non-fringe contributions.",
          "platform": [
            "Reddit"
          ],
          "targeted_group": [
            "Trans women",
            "Transgender people"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Radicalisation & violent extremism": [
              "Radicalisation pathways",
              "User migration",
              "Community growth mechanisms"
            ],
            "Spread, amplification & networks": [
              "Community infiltration",
              "Cross-platform migration"
            ],
            "Toxicity & harassment": [
              "Comment toxicity scoring"
            ]
          },
          "theme_qualifiers_canonical": {
            "Radicalisation & violent extremism": [
              "Radicalisation pathways",
              "User migration"
            ],
            "Spread, amplification & networks": [
              "Cross-platform migration"
            ],
            "Toxicity & harassment": [
              "Comment toxicity scoring",
              "Identity-targeted hate"
            ]
          },
          "tentative": false
        },
        {
          "title": "The AI Alignment Paradox",
          "wid": "the-ai-alignment-paradox",
          "type": "publication",
          "year": 2024,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2405.20806",
          "authors": [
            "Robert West",
            "Roland Aydin"
          ],
          "epfl_authors": [
            "Robert West"
          ],
          "about": "Perspective article on what it terms the AI alignment paradox: the better AI models are aligned with a set of values, the easier it may become for adversaries to realign them with opposing values. It sketches three incarnations for language models: model tinkering (for instance steering internal states), input tinkering through jailbreak prompts such as persona attacks, and output tinkering with a value editor model that minimally edits aligned outputs. Its example value edit turns a statement that Putin initiated a military operation in Ukraine into one saying he was provoked into a special operation; an autocratic state could add such a step to a wrapper website for a blocked chatbot.",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Adaptive attacks",
            "Adversarial robustness",
            "Safety alignment"
          ],
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "Prevention",
          "key_terms": [
            "AI alignment paradox",
            "Adversarial misalignment",
            "Jailbreak attacks",
            "Model steering",
            "Value editing"
          ],
          "models": [
            "ChatGPT",
            "Claude",
            "GPT-3",
            "GPT-4"
          ],
          "method_qualifiers": [
            "Jailbreaking",
            "Activation steering",
            "Adversarial evasion"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "No empirical data (perspective article).",
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Adversarial robustness",
              "Adaptive attacks",
              "Safety alignment"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Adaptive attacks",
              "Adversarial robustness",
              "Safety alignment"
            ]
          },
          "tentative": false
        },
        {
          "title": "The Amplification Paradox in Recommender Systems",
          "wid": "the-amplification-paradox-in-recommender-systems",
          "type": "publication",
          "year": 2023,
          "venue": "ICWSM 2023",
          "link": "https://arxiv.org/abs/2302.11225",
          "authors": [
            "Manoel Horta Ribeiro",
            "Veniamin Veselovsky",
            "Robert West"
          ],
          "epfl_authors": [
            "Manoel Horta Ribeiro",
            "Veniamin Veselovsky",
            "Robert West"
          ],
          "about": "Paper explaining the amplification paradox: audits found that blindly following recommendations leads users to increasingly partisan, conspiratorial or false content, yet real user traces suggest recommender systems are not the primary driver of attention toward extreme content. Simulations with a simple agent-based model of a collaborative-filtering recommender and five political topics offer a possible explanation: simulated users rarely consume niche content when given the option because it is of low utility to them, which can lead the recommender to deamplify it. The results call for a nuanced interpretation of algorithmic amplification and for modeling the utility of content to users when auditing recommender systems.",
          "themes": [
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Content nicheness",
            "Exposure intensity",
            "Recommender amplification"
          ],
          "mdh_focus": [
            "M"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "Monitoring",
          "key_terms": [
            "Algorithmic amplification",
            "Extreme content",
            "Collaborative filtering",
            "Recommender systems",
            "Agent-based model"
          ],
          "models": [],
          "method_qualifiers": [
            "Agent-based simulation",
            "Collaborative filtering"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "amplification paradox",
              "kind": "tool"
            }
          ],
          "data_description": "Synthetic simulation: 600 users, 600 items, 5 political topics.",
          "platform": [
            "YouTube"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Spread, amplification & networks": [
              "Recommender amplification",
              "Content nicheness",
              "Exposure intensity"
            ]
          },
          "theme_qualifiers_canonical": {
            "Spread, amplification & networks": [
              "Exposure intensity",
              "Recommender amplification"
            ]
          },
          "tentative": false
        },
        {
          "title": "Tube2Vec: Social and Semantic Embeddings of YouTube Channels",
          "wid": "tube2vec-social-and-semantic-embeddings-of-youtube-chan",
          "type": "publication",
          "year": 2023,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2306.17298",
          "authors": [
            "Léopaul Boesinger",
            "Manoel Horta Ribeiro",
            "Veniamin Veselovsky",
            "Robert West"
          ],
          "epfl_authors": [
            "Léopaul Boesinger",
            "Manoel Horta Ribeiro",
            "Veniamin Veselovsky",
            "Robert West"
          ],
          "about": "Paper building latent representations (embeddings) of YouTube channels as an alternative to manual annotation and low-recall keyword search when studying the social and semantic dimensions of channels. From YouTube links shared on Reddit between 2010 and 2022, it creates embeddings based on social sharing behavior, video metadata such as titles and descriptions, and YouTube's video recommendations, evaluated with crowdsourcing and existing datasets. Recommendation embeddings excel at capturing both social and semantic dimensions, although social-sharing embeddings correlate better with existing partisan scores. The embeddings for 44000 YouTube channels are shared for future research.",
          "themes": [
            "NA"
          ],
          "subtopics": [],
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_relevance": "infrastructure",
          "relevance": 2,
          "stage": "Monitoring",
          "key_terms": [
            "YouTube channel embeddings",
            "Social dimensions",
            "Recommendation graph",
            "Semantic similarity",
            "Computational social science"
          ],
          "models": [
            "all-MiniLM-L6-v2"
          ],
          "method_qualifiers": [
            "Graph embedding",
            "Sentence embeddings",
            "Random forest probing",
            "Supervised classification"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Tube2Vec embeddings",
              "kind": "dataset"
            },
            {
              "name": "YouTube channel embeddings (44K channels)",
              "kind": "dataset"
            }
          ],
          "data_description": "44000 YouTube channels, 77.4M Reddit tuples, 2010-2022.",
          "platform": [
            "Reddit",
            "YouTube"
          ],
          "label_version": "v5",
          "theme_qualifiers": {},
          "theme_qualifiers_canonical": {},
          "tentative": false
        }
      ]
    },
    {
      "id": "touradj",
      "name": "Touradj Ebrahimi",
      "url": "https://people.epfl.ch/touradj.ebrahimi",
      "unit": "MMSPG",
      "faculty": "STI",
      "mdh_focus": [
        "M",
        "D"
      ],
      "dataTypes": [
        "Image",
        "Video",
        "Audio",
        "Text"
      ],
      "techTypes": [
        "Provenance Standards",
        "Watermarking",
        "Deepfake Detection",
        "Media Forensics",
        "Content Authenticity",
        "Trust Indicators"
      ],
      "stage": "Prevention + Monitoring + Mitigation",
      "publications": [
        {
          "title": "An International Standard For Assessing Trustworthiness In Media",
          "wid": "an-international-standard-for-assessing-trustworthiness",
          "type": "publication",
          "year": 2024,
          "venue": "2024 IEEE International Conference on Image Processing (ICIP)",
          "link": "https://ieeexplore.ieee.org/document/10647585",
          "authors": [
            "Deepayan Bhowmik (Newcastle University)",
            "Sabrina Caldwell (Australian National University)",
            "Jaime Delgado (UPC BarcelonaTECH)",
            "Touradj Ebrahimi (EPFL / RayShaper SA)",
            "Nikolaos Fotos (UPC BarcelonaTECH)",
            "Xiaojun Gu (Huawei)",
            "Ziyuan Hu (Huawei)",
            "Xin Kang (Huawei)",
            "Fernando Pereira (Instituto de Telecomunicacoes)",
            "Leonard Rosenthol (Adobe)",
            "Frederik Temmermans (Vrije Universiteit Brussel / imec)",
            "Haibo Zhou (Huawei)"
          ],
          "epfl_authors": [
            "Touradj Ebrahimi (EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D"
          ],
          "stage": "Prevention + Monitoring",
          "relevance": 4,
          "about": "Describes the JPEG Trust framework (ISO/IEC 21617), an international standard establishing trust across digital media creation, modification, distribution and consumption. It cryptographically links provenance information to the media asset so consumers can assess trustworthiness and detect undeclared manipulation. Motivated by generative-AI synthetic media, fake-media distribution and misinformation, it standardises protocols to extract trust indicators, annotate provenance, and securely bind assets to their annotations, extending the C2PA content-authenticity approach.",
          "mdh_topics": [
            "JPEG Trust",
            "ISO/IEC 21617",
            "Media provenance",
            "Content authenticity",
            "C2PA",
            "Synthetic media"
          ],
          "themes": [
            "Platform governance & regulation",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "AI-generated content disclosure",
            "Interoperability requirements",
            "Media provenance",
            "Technical standards",
            "Trust indicators"
          ],
          "key_terms": [
            "JPEG Trust",
            "C2PA",
            "Media provenance",
            "JUMBF",
            "AI-generated content"
          ],
          "models": [],
          "method_qualifiers": [
            "Cryptographic content binding",
            "Standards requirements elicitation"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "JPEG Trust (ISO/IEC 21617) Core Foundation",
              "kind": "standard"
            },
            {
              "name": "JPEG Trust",
              "kind": "standard",
              "url": "https://jpeg.org/jpegtrust/",
              "evidence": "Paper, Section 1: \"the Joint Photographic Experts Group (JPEG) committee (ISO/IEC JTC 1/SC 29/WG 1) has initiated a new international standard, JPEG Trust (ISO/IEC 21617), which is expected to be published in 2024\", with footnote 1 pointing at \"Documentation on JPEG Trust: Workshop Proceedings https://jpeg.org/jpegtrust/documentation.html\". The standard's own page states: \"JPEG Trust (ISO/IEC 21617) provides a framework for establishing trust in media.\" and \"JPEG Trust Part 1 Core Foundation specifies aspects of authenticity, provenance, attribution, intellectual property rights, and integrity through secure and reliable annotation of the media assets throughout their life cycle.\""
            }
          ],
          "follow_up": [
            {
              "what": "ISO/IEC 21617-1 (JPEG Trust Part 1, Core Foundation) was approved and published as an International Standard in January 2025, and a second edition is now in development. The paper describes it while it was still a Committee Draft.",
              "kind": "standard",
              "url": "https://jpeg.org/jpegtrust/",
              "evidence": "\"JPEG Trust Part 1 Core Foundation specifies aspects of authenticity, provenance, attribution, intellectual property rights, and integrity through secure and reliable annotation of the media assets throughout their life cycle. This part was published in January 2025, a second edition of this part is currently under development.\" The JPEG committee press release of 2 December 2024 states: \"the first part of JPEG Trust, the 'Core Foundation' (ISO/IEC IS 21617-1) International Standard, has now been approved\" (https://jpeg.org/items/20241202_press.html)."
            },
            {
              "what": "JPEG Trust Part 2, Trust Profiles Catalogue - a catalogue of reusable snippets for constructing the Trust Profiles this paper defines, now under development",
              "kind": "standard",
              "url": "https://jpeg.org/jpegtrust/",
              "evidence": "\"JPEG Trust Part 2 provides a catalogue of snippets that can be used for the purpose of constructing Trust Profiles, which can be used for assessing the trustworthiness of media assets in given usage scenarios. This Part is currently under development.\""
            },
            {
              "what": "JPEG Trust Part 3, Media Asset Watermarking - extends the framework with watermarking as a binding and labelling mechanism, under development",
              "kind": "standard",
              "url": "https://jpeg.org/jpegtrust/",
              "evidence": "\"JPEG Trust Part 3 defines the use of watermarking as one of the components of the JPEG Trust framework to support tools and mechanisms for content authenticity, provenance, integrity, labelling, and binding between JPEG Trust metadata and corresponding media assets. This Part is currently under development.\" The 2 December 2024 press release adds: \"JPEG Trust has an ambitious schedule of future work, including evolving and extending the core foundation into related topics of media tokenization and media asset watermarking\"."
            },
            {
              "what": "JPEG Trust Part 4, Reference Software - reference implementations of the framework, under development",
              "kind": "standard",
              "url": "https://jpeg.org/jpegtrust/",
              "evidence": "\"JPEG Trust Part 4 provides a set of JPEG Trust reference software implementations. This Part is currently under development.\""
            },
            {
              "what": "JPEG Trust Watermarking Benchmark, an ICIP 2026 Grand Challenge co-organised by Touradj Ebrahimi (EPFL) with three of this paper's other co-authors, evaluating watermarking algorithms against the criteria of JPEG Trust Part 3",
              "kind": "benchmark",
              "url": "https://jpeg-trust-community.github.io/watermarking/benchmark/index.html",
              "evidence": "\"This grand challenge aims to assess watermarking performance ... along various evaluation criteria set out by the JPEG Trust Part 3: Media Asset Watermarking\" initiative, describing \"JPEG Trust (ISO/IEC 21617)\" as \"an international standardisation effort that provides a framework for establishing trust in media.\" Organisers listed: \"Dr Deepayan Bhowmik - Newcastle University, UK\", \"Prof Touradj Ebrahimi - EPFL, Switzerland\" (\"the current Convener of the JPEG standardisation Committee\", \"one of the key editors of the JPEG Trust international standard\"), \"Dr Sabrina Caldwell\", \"Dr Frederik Temmermans - Vrije Universiteit Brussel & imec, Belgium\"."
            }
          ],
          "data_description": "No empirical data (standards proposal).",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "Platform governance & regulation": [
              "Technical standards",
              "Interoperability requirements"
            ],
            "Verification & content authenticity": [
              "Media provenance",
              "Trust indicators",
              "AI-generated content disclosure"
            ]
          },
          "theme_qualifiers_canonical": {
            "Platform governance & regulation": [
              "Interoperability requirements",
              "Technical standards"
            ],
            "Verification & content authenticity": [
              "AI-generated content disclosure",
              "Media provenance",
              "Trust indicators"
            ]
          },
          "tentative": false
        },
        {
          "title": "Assessment Framework for Deepfake Detection in Real-world Situations",
          "wid": "assessment-framework-for-deepfake-detection-in-real-wor",
          "type": "publication",
          "year": 2023,
          "link": "https://arxiv.org/abs/2304.06125",
          "authors": [
            "Yuhang Lu",
            "Touradj Ebrahimi"
          ],
          "epfl_authors": [
            "Yuhang Lu",
            "Touradj Ebrahimi"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Deepfakes",
            "Deepfake detection",
            "Robustness benchmark",
            "Data augmentation",
            "Media forensics",
            "FaceForensics++"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "Study building the first systematic way to test how well learning-based deepfake detectors hold up under realistic conditions rather than only on clean benchmark data. It runs a controlled battery of image and video degradations (noise, resizing, compression, blurring, brightness and contrast changes) at several severity levels, then re-tests each detector on the degraded copies. Detectors that look near-perfect on clean data collapse under mild processing: Capsule-Forensics drops from 99.20 percent AUC on unaltered images to 53.55 percent under deep-learning-based compression. The authors propose a stochastic degradation-based augmentation method that lifts overall AUC back to 86.16 percent while keeping clean-data performance high.",
          "why": "It is a deepfake-detection robustness study showing detectors fail on real-world degraded media and proposing an augmentation that recovers most of the lost accuracy.",
          "data": "Image, Video",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Deepfake detection",
            "Detector robustness"
          ],
          "key_terms": [
            "Deepfake detection",
            "Face manipulation",
            "Data augmentation",
            "Real-world distortions",
            "Celeb-DF"
          ],
          "models": [
            "Capsule-Forensics",
            "EfficientNet-b4",
            "SBIs",
            "XceptionNet"
          ],
          "method_qualifiers": [
            "Data augmentation",
            "Cross-dataset evaluation"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Assessment framework for deepfake detection in real-world situations",
              "kind": "benchmark"
            },
            {
              "name": "Deepfake detection assessment framework (Python toolbox)",
              "kind": "tool"
            },
            {
              "name": "Stochastic Degradation-based Augmentation (SDAug)",
              "kind": "framework"
            },
            {
              "name": "Deepfake detection assessment toolbox",
              "kind": "tool",
              "evidence": "Paper (arXiv 2304.06125, sec. 1): \"A flexible Python toolbox is developed and the source code of the proposed assessment framework is released to facilitate relevant research activities.\" No repository URL is given anywhere in the preprint or in the peer-reviewed journal version, and none was found online."
            },
            {
              "name": "SDAug",
              "kind": "framework",
              "evidence": "Paper (arXiv 2304.06125, sec. 1): \"Inspired by the real-world data degradation process, a stochastic degradationbased augmentation (SDAug) method driven by typical image and video processing operations is designed for deepfake detection tasks.\" No public code release found."
            }
          ],
          "follow_up": [
            {
              "what": "Peer-reviewed journal version: Lu & Ebrahimi, \"Assessment framework for deepfake detection in real-world situations\", EURASIP Journal on Image and Video Processing 2024, article 6 (published 13 February 2024). Same title and authors as the preprint, extended from three to four evaluated detectors (Capsule-Forensics, XceptionNet, SBIs, UIA-VIT).",
              "kind": "successor-work",
              "url": "https://doi.org/10.1186/s13640-024-00621-8",
              "evidence": "Journal abstract, verbatim: \"To demonstrate the effectiveness and usage of the framework, extensive experiments and detailed analysis of four popular deepfake detection methods are further presented in this paper. In addition, a stochastic degradation-based data augmentation method driven by realistic processing operations is designed, which significantly improves the robustness of deepfake detectors.\" The arXiv version of the same sentence reads \"three popular deepfake detection methods\"."
            },
            {
              "what": "DAPS (degradation-based amplitude-phase switch) augmentation, Lu, Luo & Ebrahimi, SPIE Applications of Digital Image Processing XLVI, vol. 12674 (2023). A later MMSPG augmentation method whose evaluation is run on this work's assessment framework, which it cites as reference 56.",
              "kind": "successor-work",
              "url": "https://doi.org/10.1117/12.2676400",
              "evidence": "DAPS paper sec. 4, verbatim: \"Therefore, a realistic image and video assessment framework56 has been employed for a fair measurement and comparison among different augmentation methods.\" Its reference [56] is: \"Lu, Y. and Ebrahimi, T., 'Assessment framework for deepfake detection in real-world situations,' arXiv preprint arXiv:2304.06125 (2023).\""
            }
          ],
          "data_description": "FaceForensics++ and Celeb-DFv2 datasets, image and video deepfakes.",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness"
            ]
          },
          "tentative": false
        },
        {
          "title": "JPEG Trust, ISO/IEC 21617 (Core Foundation, 2025) International Standard",
          "wid": "jpeg-trust-iso-iec-21617-core-foundation-2025-internati",
          "type": "project",
          "year": 2025,
          "link": "https://jpeg.org/jpegtrust/",
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "JPEG Trust",
            "International standard",
            "Media provenance",
            "Deepfakes",
            "Content authenticity",
            "ISO/IEC"
          ],
          "stage": "NA",
          "relevance": 4,
          "about": "International standard, ISO/IEC 21617, defining the core foundation of JPEG Trust for media provenance and authenticity. It gives media files signals about their origin and how they were produced, without issuing a centralised verdict on whether content is true. It is positioned to address a reported 500 billion dollar per year cost of disinformation.",
          "why": "A flagship international standard for media provenance and authenticity built specifically to counter mis- and disinformative media.",
          "data": "not applicable (standard)",
          "what": "International standard that gives media assets a way to carry trust signals across their whole life cycle, so anyone who encounters an image, video, or audio file can judge how much to trust it. The annotations cover authenticity (has the content been altered), provenance (where it came from and how it was made), attribution, intellectual property rights, and integrity. It is built in parts spanning the core annotation framework, configurable trust profiles, watermarking that binds trust metadata to the content so it survives transformations, and reference software. Built to work across the JPEG family of formats and extendable to video and audio, it aims to counter mis- and disinformative media by giving the ecosystem tools to signal trustworthiness without forcing a single centralized verdict on what is true or false. The standard is currently structured in four parts. Part 1, the core foundation, was published in January 2025 and defines the annotation framework itself. Part 2 provides a catalogue of trust profiles, which are essentially configurable sets of trust criteria that can be applied to assess a media asset in a given context. Part 3 defines how watermarking can be used as one of the components of this framework, providing a mechanism that binds trust metadata to the content itself in a way that survives transformations like transcoding or rescanning. Part 4 will provide reference software implementations. The standard is designed to integrate into existing digital media ecosystems and is compatible with the full JPEG family of formats, but its generic architecture means it can also be applied to video and audio. The motivation is that the global cost of mis- and disinformative media is estimated at over US$500 billion per year, and the standard is intended to give the media ecosystem the tools to address that without requiring a single centralized verdict on what is true or false.\nThe standard is currently structured in four parts. Part 1, the core foundation, was published in January 2025 and defines the annotation framework itself. Part 2 provides a catalogue of trust profiles, which are essentially configurable sets of trust criteria that can be applied to assess a media asset in a given context. Part 3 defines how watermarking can be used as one of the components of this framework, providing a mechanism that binds trust metadata to the content itself in a way that survives transformations like transcoding or rescanning. Part 4 will provide reference software implementations.\n  \nThe standard is designed to integrate into existing digital media ecosystems and is compatible with the full JPEG family of formats, but its generic architecture means it can also be applied to video and audio. The motivation sis that the global cost of mis- and disinformative media is estimated at over US$500 billion per year, and the standard is intended to give the media ecosystem the tools to address that without requiring a single centralized verdict on what is true or false.",
          "themes": [
            "Platform governance & regulation",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Interoperability requirements",
            "Lifecycle annotation",
            "Media provenance",
            "Technical standards"
          ],
          "key_terms": [
            "JPEG Trust",
            "Media provenance",
            "Content authenticity",
            "ISO/IEC 21617",
            "Technical standards"
          ],
          "models": [],
          "method_qualifiers": [],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "JPEG Trust (ISO/IEC 21617)",
              "kind": "standard"
            },
            {
              "name": "JPEG Trust (ISO/IEC 21617-1 Core Foundation)",
              "kind": "standard",
              "url": "https://www.iso.org/standard/86831.html",
              "evidence": "ISO/IEC 21617-1:2025 Information technology - JPEG Trust Part 1: Core foundation. Abstract: This document specifies a framework for establishing trust in media. This framework includes aspects of authenticity, provenance and integrity through secure and reliable annotation of the media assets throughout their life cycle. General information Status : Withdrawn Publication date : 2025-01 Edition : 1 Number of pages : 55 Technical Committee : ISO/IEC JTC 1/SC 29"
            }
          ],
          "follow_up": [
            {
              "what": "Second edition of the same standard, ISO/IEC 21617-1:2026, published August 2026; it replaces and withdraws the 2025 first edition and widens the framework to attribution and intellectual property rights",
              "kind": "standard",
              "url": "https://www.iso.org/standard/91405.html",
              "evidence": "ISO/IEC 21617-1:2026 Information technology - JPEG Trust Part 1: Core foundation. Published (Edition 2, 2026). Publication date : 2026-08. Edition : 2. Number of pages : 73. Life cycle: Previously Withdrawn ISO/IEC 21617-1:2025. Now Published ISO/IEC 21617-1:2026."
            },
            {
              "what": "JPEG Trust Part 2, Trust profiles and reports, built directly on the Trust Indicators specified in Part 1 (ISO/IEC DIS 21617-2, at DIS enquiry stage 40.20)",
              "kind": "standard",
              "url": "https://www.iso.org/standard/89038.html",
              "evidence": "ISO/IEC DIS 21617-2 Information technology - JPEG Trust Part 2: Trust profiles and reports. Abstract: The scope of JPEG Trust Part 2 is to elaborate on Trust Profiles and Trust Reports, which are built on the Trust Indicators specified in JPEG Trust Part 1."
            },
            {
              "what": "JPEG Trust Part 3, Media asset watermarking, adding watermarking as a component of the JPEG Trust framework (ISO/IEC DIS 21617-3, DIS ballot initiated)",
              "kind": "standard",
              "url": "https://www.iso.org/standard/90209.html",
              "evidence": "ISO/IEC DIS 21617-3 Information technology - JPEG Trust Part 3: Media asset watermarking. Under development. Stage : DIS ballot initiated: 12 weeks [ 40.20 ]. Edition : 1. Number of pages : 20."
            },
            {
              "what": "JPEG Trust Part 4, Reference software, a set of JPEG Trust reference software implementations (ISO/IEC DIS 21617-4, at stage 40.20)",
              "kind": "standard",
              "url": "https://jpeg.org/jpegtrust/workplan.html",
              "evidence": "Under development: ISO/IEC DIS 21617-4 : Reference software (ISO page, ISO edition 1, in stage 40.20). And from the overview page: JPEG Trust Part 4 provides a set of JPEG Trust reference software implementations. This Part is currently under development."
            },
            {
              "what": "indicator-extractor, a WG1 reference CLI tool that extracts JPEG Trust Trust Indicator Sets from files per the Part 1 specification (Apache v2.0)",
              "kind": "repository",
              "url": "https://gitlab.com/wg1/jpeg-trust/indicator-extractor",
              "evidence": "From jpeg.org/jpegtrust/software.html, the JPEG Trust Software table: 'indicator-extractor | CLI tool to extract Trust Indicator Sets (JSON) from input files, per JPEG Trust spec | Node.js | Apache v2.0', linking to https://gitlab.com/wg1/jpeg-trust/indicator-extractor. The GitLab project resolves and is owned by the wg1 group."
            },
            {
              "what": "profile-evaluator, a WG1 reference CLI tool that evaluates Trust Indicator Sets against a Trust Profile, implementing JPEG Trust Parts 1 and 2 (Apache v2.0)",
              "kind": "repository",
              "url": "https://gitlab.com/wg1/jpeg-trust/profile-evaluator",
              "evidence": "From jpeg.org/jpegtrust/software.html, the JPEG Trust Software table: 'profile-evaluator | CLI tool to evaluate Trust Indicator Sets against a Trust Profile (JPEG Trust parts 1 & 2) | Node.js | Apache v2.0', linking to https://gitlab.com/wg1/jpeg-trust/profile-evaluator. The GitLab project resolves and is owned by the wg1 group."
            }
          ],
          "data_description": "No empirical data (ISO/IEC standard specification).",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "Platform governance & regulation": [
              "Technical standards",
              "Interoperability requirements"
            ],
            "Verification & content authenticity": [
              "Media provenance",
              "Lifecycle annotation"
            ]
          },
          "theme_qualifiers_canonical": {
            "Platform governance & regulation": [
              "Interoperability requirements",
              "Technical standards"
            ],
            "Verification & content authenticity": [
              "Media provenance"
            ]
          },
          "tentative": true
        },
        {
          "title": "An Introduction to the JPEG Fake Media Initiative",
          "wid": "an-introduction-to-the-jpeg-fake-media-initiative",
          "type": "publication",
          "year": 2021,
          "link": "https://ieeexplore.ieee.org/document/9565508",
          "authors": [
            "Frederik Temmermans (VUB/imec)",
            "Deepayan Bhowmik (Stirling)",
            "Fernando Pereira (IST Lisbon)",
            "Touradj Ebrahimi (MMSPG)"
          ],
          "epfl_authors": [
            "Touradj Ebrahimi (MMSPG)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "Deepfakes",
            "Media provenance",
            "JPEG standards",
            "Content authenticity",
            "Media forensics"
          ],
          "stage": "Prevention + Monitoring",
          "relevance": 4,
          "about": "Introduces the JPEG Fake Media initiative, a standards-track effort to address manipulated and synthetically generated media. It defines Fake Media broadly as any generated or modified media asset regardless of intent, since the same techniques power both legitimate creative work and serious harms, and warns that spreading manipulated media can cause social unrest, fuel rumours for political gain, or encourage hate crimes. It lays out use cases, definitions, and requirements, then derives two core technical needs: describing what modification was made and by whom, and securely linking those descriptions to the media so the metadata cannot be tampered with. It positions the effort as complementary to the Content Authenticity Initiative, with JPEG focusing on the signaling syntax.",
          "why": "It launches the standards-track effort to annotate media provenance and authenticity to counter deepfake-driven misinformation, disinformation, and incitement to hate crimes.",
          "data": "Image, Video",
          "themes": [
            "Fraud, impersonation & forgery",
            "Platform governance & regulation",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Deepfake detection",
            "Interoperability requirements",
            "Media provenance",
            "Secure metadata binding",
            "Technical standards"
          ],
          "key_terms": [
            "JPEG Fake Media",
            "Media provenance",
            "Deepfakes",
            "Content Authenticity Initiative",
            "Media forensics"
          ],
          "models": [],
          "method_qualifiers": [
            "Standards requirements elicitation"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "JPEG Fake Media",
              "kind": "standard"
            }
          ],
          "follow_up": [
            {
              "what": "Final Call for Proposals for JPEG Fake Media, issued at the 95th JPEG meeting (April 2022) - the exact next roadmap step this paper announced",
              "kind": "standard",
              "url": "https://jpeg.org/items/20220502_jpeg_fake_media_cfp.html",
              "evidence": "At its 95th online meeting, the JPEG committee released a Final Call for Proposals (CfP) for JPEG Fake Media. The scope of JPEG Fake Media is the creation of a standard that can facilitate the secure and reliable annotation of media asset creation and modifications. The standard shall address use cases that are in good faith as well as those with malicious intent."
            },
            {
              "what": "JPEG Trust, the standardisation activity the JPEG Fake Media exploration turned into",
              "kind": "standard",
              "url": "https://jpeg.org/jpegtrust/documentation.html",
              "evidence": "JPEG Trust was preceeded by an exploration on JPEG Fake Media. During this exploration several workshops were organized to interact with stakeholders and identify relevant use cases and requirements for standardization. [...] The JPEG Fake Media Use Cases and Requirements and Call for Proposal documents that led to the initiation of JPEG Trust are available below."
            },
            {
              "what": "ISO/IEC 21617-1:2025, Information technology - JPEG Trust - Part 1: Core foundation, published as an International Standard",
              "kind": "standard",
              "url": "https://www.iso.org/standard/86831.html",
              "evidence": "ISO/IEC 21617-1:2025 / Information technology - JPEG Trust - Part 1: Core foundation"
            },
            {
              "what": "JPEG press release announcing JPEG Trust was sent for publication as an International Standard at the 105th meeting (Berlin, October 2024), quoting JPEG Convenor Touradj Ebrahimi",
              "kind": "standard",
              "url": "https://jpeg.org/items/20241202_press.html",
              "evidence": "The 105th JPEG meeting was held in Berlin, Germany, from October 6 to 11, 2024. During this JPEG meeting, JPEG Trust was sent for publication as an International Standard. This is a major achievement in providing standardized tools to effectively fight against the proliferation of fake media and disinformation while restoring confidence in multimedia information."
            }
          ],
          "data_description": "No empirical data (standards proposal).",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "Platform governance & regulation": [
              "Technical standards",
              "Interoperability requirements"
            ],
            "Verification & content authenticity": [
              "Media provenance",
              "Deepfake detection",
              "Secure metadata binding"
            ]
          },
          "theme_qualifiers_canonical": {
            "Platform governance & regulation": [
              "Interoperability requirements",
              "Technical standards"
            ],
            "Verification & content authenticity": [
              "Deepfake detection",
              "Media provenance"
            ]
          },
          "tentative": false
        },
        {
          "title": "JPEG Trust Watermarking Benchmark",
          "wid": "jpeg-trust-watermarking-benchmark",
          "type": "project",
          "year": "2026-ongoing",
          "link": "https://jpeg-trust-community.github.io/watermarking/benchmark/index.html",
          "authors": [
            "Deepayan Bhowmik (Newcastle)",
            "Touradj Ebrahimi (EPFL/JPEG)",
            "Sabrina Caldwell (UNSW/ANU)",
            "Frederik Temmermans (VUB/imec)"
          ],
          "epfl_authors": [
            "Touradj Ebrahimi (EPFL/JPEG)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "Watermarking benchmark",
            "JPEG Trust Part 3",
            "ICIP 2026 challenge",
            "Generative AI attacks",
            "Reproducibility"
          ],
          "stage": "NA",
          "relevance": 4,
          "about": "A grand challenge that lets teams test their image and video watermarking algorithms against the criteria of JPEG Trust Part 3 on media asset watermarking. Participants embed a watermark and are scored on embedding quality and on robustness against signal-processing, geometric, and compression attacks, plus a generative AI-based object manipulation pipeline. Because many companies claim robust watermarking but those claims are rarely tested empirically, the challenge requires the top three teams to publish reproducible code publicly. It sits within a broader push to replace marketing claims with transparent benchmarks.",
          "why": "An independent benchmark of watermarking robustness, the main provenance defence against AI-generated and manipulated media, tied to the JPEG Trust standard.",
          "data": "Images. The challenge page names no commercial generator, only \"a new generative AI-based object manipulation/editing pipeline\".",
          "themes": [
            "Platform governance & regulation",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Benchmark reliability",
            "Detector robustness",
            "Media provenance",
            "Technical standards"
          ],
          "key_terms": [
            "JPEG Trust",
            "Image watermarking",
            "AI-generated content",
            "Media provenance",
            "AI-generated content labelling"
          ],
          "models": [],
          "method_qualifiers": [
            "Blind watermark extraction"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "JPEG Trust Watermarking Benchmark",
              "kind": "benchmark",
              "url": "https://jpeg-trust-community.github.io/watermarking/benchmark/index.html",
              "evidence": "This grand challenge aims to assess watermarking performance (e.g., embedding distortion and robustness against attacks) along various evaluation criteria set out by the JPEG Trust Part 3: Media Asset Watermarking initiative. -- and on the same page: Touradj Ebrahimi is a professor of image processing at Ecole Polytechnique Federale de Lausanne (EPFL) and the current Convener of the JPEG standardisation Committee."
            }
          ],
          "follow_up": [
            {
              "what": "ISO/IEC CD 21617-3, JPEG Trust Part 3: Media asset watermarking, the standard the benchmark's evaluation criteria are drawn from and feed back into",
              "kind": "standard",
              "url": "https://www.iso.org/standard/90209.html",
              "evidence": "ISO/IEC CD 21617-3 Information technology - JPEG Trust - Part 3: Media asset watermarking. This document defines the use of watermarking as one of the components of the JPEG Trust framework to support tools and mechanisms for content authenticity, provenance, integrity, labelling, and binding between JPEG Trust metadata and corresponding media assets."
            },
            {
              "what": "JPEG-Trust-Community/watermarking GitHub repository, which hosts the benchmark site and its evaluation code (attacks, metrics, package)",
              "kind": "repository",
              "url": "https://github.com/JPEG-Trust-Community/watermarking",
              "evidence": "Repository JPEG-Trust-Community/watermarking, public, contains top-level directories 'benchmark' (holding index.html and static/) and 'evaluation_metric' (holding attacks/, metrics/, package/). The benchmark site URL path https://jpeg-trust-community.github.io/watermarking/benchmark/index.html maps exactly onto benchmark/index.html in this repo."
            },
            {
              "what": "ICIP 2026 Grand Challenge track: benchmark launched 11 Feb 2026, submissions closed 26 Apr 2026, winners announced 4 May 2026, presented at ICIP 2026 (13-17 Sept 2026)",
              "kind": "deployment",
              "url": "https://jpeg-trust-community.github.io/watermarking/benchmark/index.html",
              "evidence": "Launch: February 11, 2026; Submission Close: April 26, 2026; Winners Announced: May 4, 2026; ICIP 2026 Presentation: September 13-17, 2026. The first three teams on the leaderboard must submit their code to GitHub."
            }
          ],
          "data_description": "Mix of real and synthetic images, plus an unseen test dataset.",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "Platform governance & regulation": [
              "Technical standards"
            ],
            "Verification & content authenticity": [
              "Media provenance",
              "Detector robustness",
              "Benchmark reliability"
            ]
          },
          "theme_qualifiers_canonical": {
            "Platform governance & regulation": [
              "Technical standards"
            ],
            "Verification & content authenticity": [
              "Detector robustness",
              "Media provenance"
            ]
          },
          "tentative": true
        },
        {
          "title": "Risk Governance and the Rise of Deepfakes",
          "wid": "risk-governance-and-the-rise-of-deepfakes",
          "type": "report",
          "year": 2021,
          "venue": "Policy Brief, co-authored with Aengus Collins",
          "link": "https://infoscience.epfl.ch/entities/publication/7490b999-60ed-495f-a177-0f8c2c1f33b2",
          "authors": [
            "Aengus Collins (IRGC)",
            "Touradj Ebrahimi (MMSPG)"
          ],
          "epfl_authors": [
            "Aengus Collins (IRGC)",
            "Touradj Ebrahimi (MMSPG)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D",
            "H"
          ],
          "mdh_topics": [
            "Deepfakes",
            "Risk governance",
            "Provenance",
            "C2PA",
            "AI Act",
            "Digital literacy",
            "Liar's dividend"
          ],
          "stage": "Prevention + Monitoring + Mitigation",
          "relevance": 5,
          "about": "Policy report mapping deepfake harms across individual, organizational, and societal levels and recommending a coordinated governance response to synthetic-media disinformation. It notes that deepfake videos increased tenfold between 2018 and 2020, that the winning Facebook detector reached 65 percent accuracy, and that YouTube handles 720000 hours of uploads per day, so even 99.9 percent detection accuracy is insufficient. It cites the April 2021 draft EU AI Act transparency obligations and argues that provenance approaches outperform a detection arms race. It offers 15 recommendations spanning detection, provenance, and digital literacy.",
          "why": "It maps deepfake harms across individual, organizational, and societal levels and recommends a coordinated governance response to synthetic-media disinformation.",
          "data": "not applicable (policy report)",
          "lab": "IRGC",
          "what": "Policy-facing report that looks at deepfakes through a governance lens for policymakers rather than technical researchers. It sorts the harms into three levels: individual (personal abuse, reputational damage), organizational (fraud, extortion, and exposure for activity that relies on documentary evidence), and societal (manipulation of public opinion and erosion of democratic politics). It argues that no single fix is enough and sets out 15 recommendations spanning technology (detection and content provenance verification), legal frameworks (how defamation, harassment, and copyright apply to synthetic media), and digital literacy. It also flags a central tension: encouraging skepticism toward digital content can itself erode the public trust that democratic discourse depends on. Directly relevant to synthetic-media disinformation.",
          "themes": [
            "Fraud, impersonation & forgery",
            "Media literacy & public resilience",
            "Online sexual abuse & image-based abuse",
            "Platform governance & regulation",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Deepfake pornography",
            "Digital literacy",
            "Fabricated evidence",
            "Media provenance",
            "Synthetic voice fraud"
          ],
          "key_terms": [
            "Deepfakes",
            "Media provenance",
            "Epistemic security",
            "Risk governance",
            "C2PA"
          ],
          "models": [],
          "method_qualifiers": [],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "No empirical data (policy and risk analysis).",
          "follow_up_checked": "2026-09-02",
          "targeted_group": [
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Fraud, impersonation & forgery": [
              "Fabricated evidence",
              "Synthetic voice fraud"
            ],
            "Media literacy & public resilience": [
              "Digital literacy"
            ],
            "Online sexual abuse & image-based abuse": [
              "Deepfake pornography"
            ],
            "Platform governance & regulation": [
              "Deepfake legislation"
            ],
            "Verification & content authenticity": [
              "Media provenance"
            ]
          },
          "theme_qualifiers_canonical": {
            "Fraud, impersonation & forgery": [
              "Fabricated evidence",
              "Identity impersonation",
              "Synthetic voice fraud"
            ],
            "Media literacy & public resilience": [
              "Critical thinking support",
              "Digital literacy"
            ],
            "Online sexual abuse & image-based abuse": [
              "Deepfake pornography",
              "Harms to depicted victims"
            ],
            "Platform governance & regulation": [
              "Deepfake legislation",
              "Technical standards",
              "Transparency obligations"
            ],
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness",
              "Media provenance"
            ]
          },
          "tentative": false
        },
        {
          "title": "Exploration of Media Blockchain Technologies for JPEG Privacy and Security",
          "wid": "exploration-of-media-blockchain-technologies-for-jpeg-p",
          "type": "publication",
          "year": 2020,
          "venue": "ICIP / JPEG Workshop",
          "link": "https://infoscience.epfl.ch/entities/publication/f32e8109-aa2a-418a-bd72-e44cb32ea9af",
          "authors": [
            "Frederik Temmermans (VUB/imec)",
            "Deepayan Bhowmik (Stirling)",
            "Fernando Pereira (IST Lisbon)",
            "Touradj Ebrahimi (MMSPG)",
            "Peter Schelkens (VUB/imec)"
          ],
          "epfl_authors": [
            "Touradj Ebrahimi (MMSPG)"
          ],
          "mdh_relevance": "infrastructure",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "JPEG standards",
            "Image provenance",
            "Blockchain",
            "Watermarking",
            "Deepfakes context",
            "C2PA precursor"
          ],
          "stage": "Prevention + Monitoring",
          "relevance": 3,
          "about": "Explores media blockchain technologies as a provenance and authenticity layer for JPEG Privacy and Security, the ISO/IEC 19566-4 standard, expected at the time of writing to be published in April 2020, and a precursor to JPEG Trust. A blockchain layer logs how an image was captured and edited so that doctored images can be caught at the source. The work is motivated by doctored images and computer-generated components that give a false impression of reality and add to the problem of fake news.",
          "why": "Foundational provenance and authenticity infrastructure, a precursor to JPEG Trust, motivated by doctored images and fake news.",
          "data": "Image, Video, blockchain-anchored provenance systems",
          "what": "Conference paper exploring blockchain as a provenance layer for multimedia, creating an immutable and decentralized log of when, where, and how an image was captured and how it was modified through its workflow. It presents the scope and implementation of the JPEG Privacy and Security standard and reports on the JPEG committee's exploration of standardization needs for media blockchain applications. The authors note that images can be easily edited to give a false impression of reality, feeding the spread of fake news, which is why proving an image's origin and tracing it through processing matters. It presents no evaluation of its own. It notes that adopting blockchain for digital image integrity verification poses several challenges at technological as well as privacy related legislation levels, and that blockchain adopted to support media applications needs to be closely integrated with widely adopted standards to ensure broad interoperability. Early authenticity-infrastructure groundwork that later fed into JPEG Trust, so its link to misinformation is indirect.",
          "themes": [
            "Platform governance & regulation",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Copyright compliance",
            "Interoperability requirements",
            "Media provenance",
            "Technical standards"
          ],
          "key_terms": [
            "JPEG Privacy and Security",
            "Media blockchain",
            "Digital rights management",
            "Fake news",
            "Content authenticity"
          ],
          "models": [],
          "method_qualifiers": [
            "Standards requirements elicitation",
            "Region-of-interest encryption",
            "Distributed systems & protocols"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "JPEG Privacy and Security (ISO/IEC 19566-4)",
              "kind": "standard"
            },
            {
              "name": "JPEG Privacy and Security",
              "kind": "standard",
              "url": "https://webstore.iec.ch/en/publication/66887",
              "evidence": "ISO/IEC 19566-4:2020, 'Information technologies - JPEG systems - Part 4: Privacy and security', published 2020-03-31: 'This document specifies privacy and security features which contribute to a system layer for JPEG standards.' The paper identifies this as the standard it describes: 'JPEG Privacy and Security is a standard defined by JPEG (ISO/IEC JTC1 SC29 WG1). More specifically, it is Part 4 of the so-called JPEG Systems standard (ISO/IEC 19566-4).'"
            }
          ],
          "follow_up": [
            {
              "what": "Media Blockchain Use Cases and Requirements version 1.0 (ISO/IEC JTC 1/SC 29/WG1 N87031), released for public feedback at the 87th JPEG meeting, 25-30 April 2020, edited by the same four authors as this paper (Bhowmik, Temmermans, Pereira, Ebrahimi)",
              "kind": "standard",
              "url": "https://jpeg.org/items/20200505_media_blockchain_use_cases_and_requirements.html",
              "evidence": "'the JPEG Committee announces a call for feedback from interested stakeholders on the first public release of the use cases and requirements document' (jpeg.org announcement). The document itself (ds.jpeg.org/documents/wg1n87031-REQ-Media_Blockchain_Use_Cases_and_Requirements.pdf) carries the header 'ISO/IEC JTC 1/SC 29/WG1 N87031 / 87th Meeting, Online, 25 April-30 April 2020 / TITLE: Media Blockchain Use Cases and Requirements version 1.0 / EDITOR: Deepayan Bhowmik, Frederik Temmermans, Fernando Pereira, Touradj Ebrahimi'. The paper anticipates exactly this: 'the Committee has identified a diversified set of use cases and related requirements which are planned to be published during the next Committee meeting in April 2020.'"
            },
            {
              "what": "JPEG Fake Media exploration and its public Ad Hoc Group, created at the 88th JPEG meeting (July 2020) as the direct continuation of the media blockchain exploration described in this paper",
              "kind": "project",
              "url": "https://jpeg.org/items/20200803_fake_media.html",
              "evidence": "'The latter is closely related to issues highlighted in media blockchain under progress in the last two years in JPEG and therefore is considered as a natural continuation of that effort.' and 'During its 88th online meeting (July 2020), the JPEG Committee has created a public Ad Hoc Group (AHG) on Fake Media as a first concrete action toward the above mentioned objectives.'"
            },
            {
              "what": "JPEG Trust (ISO/IEC 21617), Part 1 Core foundation published January 2025, the standard that grew out of the JPEG Fake Media exploration",
              "kind": "standard",
              "url": "https://jpeg.org/jpegtrust/documentation.html",
              "evidence": "'JPEG Trust was preceeded by an exploration on JPEG Fake Media. During this exploration several workshops were organized to interact with stakeholders and identify relevant use cases and requirements for standardization.' The JPEG Trust overview page adds: 'JPEG Trust arose from an exploration of requirements for addressing mis- and dis-information in digital media.'"
            }
          ],
          "data_description": "No empirical data (standards description and exploration study).",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "Platform governance & regulation": [
              "Technical standards",
              "Copyright compliance",
              "Interoperability requirements"
            ],
            "Verification & content authenticity": [
              "Media provenance"
            ]
          },
          "theme_qualifiers_canonical": {
            "Platform governance & regulation": [
              "Copyright compliance",
              "Interoperability requirements",
              "Technical standards"
            ],
            "Verification & content authenticity": [
              "Media provenance"
            ]
          },
          "tentative": false
        },
        {
          "title": "A Novel Assessment Framework for Learning-based Deepfake Detectors in Realistic Conditions",
          "wid": "a-novel-assessment-framework-for-learning-based-deepfak",
          "type": "publication",
          "year": 2022,
          "venue": "SPIE Applications of Digital Image Processing XLV",
          "link": "https://doi.org/10.1117/12.2636683",
          "authors": [
            "Yuhang Lu",
            "Touradj Ebrahimi"
          ],
          "epfl_authors": [
            "Yuhang Lu",
            "Touradj Ebrahimi"
          ],
          "about": "Paper proposing a framework to assess learning-based deepfake detectors in more realistic situations, since current assessment and ranking approaches in benchmarks or competitions are unreliable. Copies of a test set are degraded with six categories of processing operations or corruptions (noise, resizing, compression, denoising, enhancement and combinations), with over five severity levels per type. Tests on Capsule-Forensics and XceptionNet show that even mild real-world processing operations can obviously harm detection accuracy, with noise and blur the most prominent factors. A stochastic degradation-based augmentation (SDAug) improves robustness and raises cross-dataset AUC on Celeb-DFv2 from 54.39 to 71.86 percent for Capsule-Forensics and from 50.00 to 73.88 percent for XceptionNet.",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Benchmark reliability",
            "Deepfake detection",
            "Detector robustness"
          ],
          "mdh_focus": [
            "D"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "Monitoring",
          "key_terms": [
            "Deepfake detection",
            "Data augmentation",
            "Face manipulation",
            "Realistic distortions",
            "Assessment framework"
          ],
          "models": [
            "Capsule-Forensics",
            "XceptionNet"
          ],
          "method_qualifiers": [
            "Data augmentation",
            "Corruption robustness evaluation"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Deepfake detection assessment framework",
              "kind": "tool"
            },
            {
              "name": "Python toolbox",
              "kind": "tool"
            },
            {
              "name": "Realistic assessment framework",
              "kind": "benchmark"
            },
            {
              "name": "SDAug",
              "kind": "framework"
            }
          ],
          "data_description": "FaceForensics++ (5000 videos) and Celeb-DFv2 (6229 videos), frame-level testing.",
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness",
              "Benchmark reliability"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness"
            ]
          },
          "tentative": false
        },
        {
          "title": "A Novel Framework for Assessment of Learning-based Detectors in Realistic Conditions with Application to Deepfake Detection",
          "wid": "a-novel-framework-for-assessment-of-learning-based-dete",
          "type": "publication",
          "year": 2022,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2203.11797",
          "authors": [
            "Yuhang Lu",
            "Ruizhi Luo",
            "Touradj Ebrahimi"
          ],
          "epfl_authors": [
            "Yuhang Lu",
            "Ruizhi Luo",
            "Touradj Ebrahimi"
          ],
          "about": "Paper proposing a framework to assess learning-based detectors in more realistic situations, since the impact of conventional distortions and processing operations such as compression, noise and enhancement is not sufficiently studied in public benchmarks. Applied to deepfake detection with Capsule-Forensics and XceptionNet on FaceForensics++ and Celeb-DF, it shows that even mild real-world processing operations can obviously harm detection accuracy, with natural noise and Gaussian blur as very prominent factors. An augmentation chain of enhancement, smoothing, Gaussian noise and JPEG compression raises cross-dataset AUC on Celeb-DF from 46.70 to 74.84 percent for Capsule-Forensics and from 50.70 to 80.67 percent for XceptionNet.",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Deepfake detection",
            "Detector robustness"
          ],
          "mdh_focus": [
            "D"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "Monitoring",
          "key_terms": [
            "Deepfake detection",
            "Data augmentation",
            "Detector robustness",
            "Assessment framework",
            "Celeb-DF"
          ],
          "models": [
            "Capsule-Forensics",
            "XceptionNet"
          ],
          "method_qualifiers": [
            "Data augmentation",
            "Cross-dataset generalisation",
            "Image corruption robustness evaluation"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Assessment framework for learning-based detectors in realistic conditions",
              "kind": "framework"
            }
          ],
          "data_description": "FaceForensics++ (1000 pristine, 4000 manipulated videos) and Celeb-DF, analysed as frames.",
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness"
            ]
          },
          "tentative": false
        },
        {
          "title": "Impact of Benign Modifications on Discriminative Performance of Deepfake Detectors",
          "wid": "impact-of-benign-modifications-on-discriminative-perfor",
          "type": "publication",
          "year": 2021,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2111.07468",
          "authors": [
            "Yuhang Lu",
            "Evgeniy Upenik",
            "Touradj Ebrahimi"
          ],
          "epfl_authors": [
            "Yuhang Lu",
            "Evgeniy Upenik",
            "Touradj Ebrahimi"
          ],
          "about": "Paper proposing a framework to assess deepfake detectors under benign processing that videos and images on the Internet and social networks constantly undergo, such as compression, denoising, enhancement and resizing, whose impact it says is not sufficiently studied. Applying conventional and learning-based operations to the FaceForensics++ dataset and testing the Capsule-Forensics detector, it finds that even benign operations generally cause a noticeable decline in detection performance. Accuracy fell from 80.25 percent on raw videos to 56.61 percent with additive Gaussian noise of variance 0.01 and 68.23 percent with libx264 compression at CRF 40; overall, combining operations degraded it further, while linear-interpolation resizing with a scale of 1.3 improved it.",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Benchmark reliability",
            "Deepfake detection",
            "Detector robustness"
          ],
          "mdh_focus": [
            "D"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "Monitoring",
          "key_terms": [
            "Deepfake detection",
            "Detector robustness",
            "FaceForensics++",
            "Benign post-processing",
            "Benign modifications"
          ],
          "models": [
            "Capsule-Forensics",
            "VGG19"
          ],
          "method_qualifiers": [
            "Post-processing robustness"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "140 test videos (10 frames each) from FaceForensics++ dataset.",
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Detector robustness",
              "Deepfake detection",
              "Benchmark reliability"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness"
            ]
          },
          "tentative": false
        },
        {
          "title": "Impact of Video Processing Operations in Deepfake Detection",
          "wid": "impact-of-video-processing-operations-in-deepfake-detec",
          "type": "publication",
          "year": 2023,
          "venue": "24th International Conference on Digital Signal Processing (DSP 2023)",
          "link": "https://doi.org/10.1109/dsp58604.2023.10167906",
          "authors": [
            "Yuhang Lu",
            "Touradj Ebrahimi"
          ],
          "epfl_authors": [
            "Yuhang Lu",
            "Touradj Ebrahimi"
          ],
          "about": "Study of how video processing operations common on social media affect deep learning deepfake detectors, which are often evaluated on benchmarks that hardly reflect real-world situations. It proposes an assessment method that applies each of seven categories of operations (compression, flipping, video filters, brightness, contrast, noise and resolution) to an entire copy of the test set, then evaluates three detectors, CapsuleNet, XceptionNet and SBIs, on FaceForensics++. None of the three accurately classifies deepfakes processed by heavy compression, resolution reduction or video noise, while vertical flipping affects XceptionNet and CapsuleNet but has limited impact on SBIs.",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Deepfake detection",
            "Detector robustness",
            "Realistic evaluation"
          ],
          "mdh_focus": [
            "D"
          ],
          "mdh_relevance": "direct",
          "relevance": 5,
          "stage": "Monitoring",
          "key_terms": [
            "Deepfake detection",
            "Video processing operations",
            "Detector robustness",
            "FaceForensics++",
            "Benchmarking"
          ],
          "models": [
            "CapsuleNet",
            "SBIs",
            "XceptionNet"
          ],
          "method_qualifiers": [
            "Perturbation robustness testing"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "4000 manipulated and 1000 pristine videos from FaceForensics++ dataset.",
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Detector robustness",
              "Deepfake detection",
              "Realistic evaluation"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness"
            ]
          },
          "tentative": false
        },
        {
          "title": "Improving Deepfake Detectors against Real-world Perturbations with Amplitude-Phase Switch Augmentation",
          "wid": "improving-deepfake-detectors-against-real-world-perturb",
          "type": "publication",
          "year": 2023,
          "venue": "SPIE Applications of Digital Image Processing XLVI",
          "link": "https://doi.org/10.1117/12.2676400",
          "authors": [
            "Yuhang Lu",
            "Ruizhi Luo",
            "Touradj Ebrahimi"
          ],
          "epfl_authors": [
            "Yuhang Lu",
            "Ruizhi Luo",
            "Touradj Ebrahimi"
          ],
          "about": "Paper proposing DAPS (degradation-based amplitude-phase switch), a data augmentation method to make deepfake detectors robust to real-world perturbations such as resizing and compression. It applies a chain of simulated real-world degradations to a training image, then recombines the degraded image's amplitude spectrum with the original's phase spectrum, so the detector focuses on the more resilient phase. With XceptionNet and UIA-VIT trained on FaceForensics++, DAPS scores considerably higher than classical augmentation techniques under most image perturbations, and in video tests it outperforms other approaches particularly under heavy compression, low resolution and temporal noise.",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Deepfake detection",
            "Detector robustness",
            "Realistic benchmarking"
          ],
          "mdh_focus": [
            "D"
          ],
          "mdh_relevance": "direct",
          "relevance": 5,
          "stage": "Monitoring",
          "key_terms": [
            "Deepfake detection",
            "Real-world perturbations",
            "Phase spectrum",
            "Data augmentation",
            "Frequency domain"
          ],
          "models": [
            "UIA-VIT",
            "XceptionNet"
          ],
          "method_qualifiers": [
            "Data augmentation",
            "Frequency domain analysis",
            "Amplitude-phase switch"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "FaceForensics++ dataset, 1000 videos, 4 manipulation types, frames extracted.",
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness",
              "Realistic benchmarking"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness"
            ]
          },
          "tentative": false
        },
        {
          "title": "Towards the Detection of AI-Synthesized Human Face Images",
          "wid": "towards-the-detection-of-ai-synthesized-human-face-imag",
          "type": "publication",
          "year": 2024,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2402.08750",
          "authors": [
            "Yuhang Lu",
            "Touradj Ebrahimi"
          ],
          "epfl_authors": [
            "Yuhang Lu",
            "Touradj Ebrahimi"
          ],
          "about": "Benchmark for detecting entirely AI-synthesized human face images, which can be abused to create fake profiles for fraud or to spread misinformation. It combines real CelebA-HQ images with faces generated by three GANs (ProGAN, StyleGAN2, VQGAN) and four diffusion models (DDPM, DDIM, PNDM, LDM) to evaluate the generalization and robustness of detectors. Detectors trained only on general categories of fake images struggle with synthetic faces, and generalization across models and robustness to perturbations remain challenges for most methods. A frequency domain analysis finds notable discrepancies between real and synthetic face spectra, and training detectors on frequency representations can significantly enhance their performance and generalization.",
          "themes": [
            "Fraud, impersonation & forgery",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "AI-generated fake profiles",
            "Cross-forgery generalisation",
            "Deepfake detection",
            "Detector robustness",
            "Fabricated evidence",
            "Identity impersonation"
          ],
          "mdh_focus": [
            "D"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "Monitoring",
          "key_terms": [
            "Deepfake detection",
            "Diffusion models",
            "Frequency analysis",
            "Synthetic face images",
            "AI-synthesized face images"
          ],
          "models": [
            "DDIM",
            "DDPM",
            "EfficientNetB4",
            "Grag2021",
            "LDM",
            "Mandelli2022",
            "Ojha2023",
            "PNDM",
            "ProGAN",
            "ResNet-50",
            "StyleGAN2",
            "VQGAN",
            "Wang2020",
            "XceptionNet"
          ],
          "method_qualifiers": [
            "Frequency domain analysis",
            "Cross-dataset generalisation"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "A Dataset of Synthetic Face Images",
              "kind": "dataset"
            },
            {
              "name": "Detection benchmark for synthetic human face images",
              "kind": "benchmark"
            }
          ],
          "data_description": "40k images per model, 7 generative models, CelebA-HQ real images.",
          "label_version": "v5",
          "theme_qualifiers": {
            "Fraud, impersonation & forgery": [
              "Fabricated evidence",
              "Identity impersonation",
              "AI-generated fake profiles"
            ],
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness",
              "Cross-forgery generalisation"
            ]
          },
          "theme_qualifiers_canonical": {
            "Fraud, impersonation & forgery": [
              "Fabricated evidence",
              "Identity impersonation"
            ],
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "susstrunk",
      "name": "Sabine Süsstrunk",
      "url": "https://people.epfl.ch/sabine.susstrunk",
      "unit": "IVRL",
      "faculty": "IC (Dean)",
      "mdh_focus": [
        "D"
      ],
      "dataTypes": [
        "Image",
        "Video"
      ],
      "techTypes": [
        "Deepfake Detection",
        "Video Forensics",
        "Computer Vision",
        "Media Literacy"
      ],
      "stage": "Prevention + Monitoring",
      "publications": [
        {
          "title": "Deepfake Awareness Booth",
          "wid": "deepfake-awareness-booth",
          "type": "event",
          "year": "2021-2025",
          "venue": "IC Open House, later WEF Davos and other EPFL outreach events",
          "link": "https://actu.epfl.ch/news/deepfake-booth-at-ai-for-good-and-ehf-2025/",
          "authors": [
            "Quentin Bammey (IVRL)",
            "Gaël Hurlimann (Mediacom)",
            "Paul Madélénat (Mediacom)",
            "Philippe Stoll (ICRC)"
          ],
          "epfl_authors": [
            "Quentin Bammey (IVRL)",
            "Gaël Hurlimann (Mediacom)",
            "Paul Madélénat (Mediacom)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Deepfake awareness",
            "Public outreach",
            "Media literacy",
            "Pedagogy"
          ],
          "stage": "Prevention",
          "relevance": 4,
          "about": "Immersive, interactive booth on deepfakes, developed by IVRL (Quentin Bammey) with EPFL Mediacom and the International Committee of the Red Cross, in which participants \"can dive into a world of deepfakes and experience what it means to face them\". It was presented at the AI for Good summit and the European Humanitarian Forum in 2025. In Süsstrunk's framing from the interview, it works because it is honest about probabilism, shows the failure mode rather than arbitrating truth, and lets a non-expert audience feel the asymmetry between how easy fabrication is and how hard verification is.",
          "why": "Public-facing awareness work on synthetic media: in the organisers' words, \"to raise awareness of the real-life consequences of deepfakes\", which they list as disinformation, fraud and scams.",
          "data": "not applicable (public awareness event)",
          "themes": [
            "Fraud, impersonation & forgery",
            "Media literacy & public resilience"
          ],
          "subtopics": [
            "Deepfake-enabled scams",
            "Immersive demonstration",
            "Public awareness campaigns"
          ],
          "key_terms": [
            "Deepfakes",
            "Generative AI",
            "Public awareness",
            "Humanitarian outreach",
            "AI for Good"
          ],
          "models": [],
          "method_qualifiers": [],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Deepfake Booth",
              "kind": "platform",
              "url": "https://actu.epfl.ch/news/deepfake-booth-at-ai-for-good-and-ehf-2025/",
              "evidence": "Official EPFL page for the artefact, titled \"Deepfake Booth at AI for Good and EHF 2025\": the booth is \"an immersive, interactive environment, participants can dive into a world of deepfakes and experience what it means to face them\", built by Dr. Quentin Bammey (IVRL) with Gael Hurlimann and Paul Madelenat (Mediacom) in collaboration with the International Committee of the Red Cross. IVRL is Sabine Susstrunk's laboratory."
            }
          ],
          "data_description": "No empirical data (outreach event description).",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "Fraud, impersonation & forgery": [
              "Deepfake-enabled scams"
            ],
            "Media literacy & public resilience": [
              "Immersive demonstration",
              "Public awareness campaigns"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media literacy & public resilience": [
              "Public awareness campaigns"
            ]
          },
          "tentative": true
        },
        {
          "title": "Efficient Temporally-Aware DeepFake Detection using H.264 Motion Vectors",
          "wid": "efficient-temporally-aware-deepfake-detection-using-h-2",
          "type": "publication",
          "year": 2024,
          "venue": "IS&T Electronic Imaging, Media Watermarking, Security, and Forensics (MWSF) 2024",
          "link": "https://arxiv.org/abs/2311.10788",
          "authors": [
            "Peter Grönquist",
            "Yufan Ren",
            "Qingyi He",
            "Alessio Verardo",
            "Sabine Süsstrunk"
          ],
          "epfl_authors": [
            "Sabine Süsstrunk"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D",
            "H"
          ],
          "mdh_topics": [
            "Deepfake detection",
            "Video forensics",
            "Temporal artefacts",
            "Real-time detection",
            "H.264"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "Deepfake video detector built for speed by reusing the motion data the H.264 codec already computes. Instead of running optical flow to catch unnatural movement between frames, it pulls the codec's own motion vectors, crops them to the detected face region, and feeds them into a lightweight network. Producing the motion-vector input costs orders of magnitude fewer operations than optical flow while matching or beating it on temporal accuracy, and combined RGB plus motion-vector input reaches around 96 percent accuracy with better generalization across unseen forgery types.",
          "why": "A cheap temporal detector for manipulated faces, which the authors say could lead to real-time detection in video calls and streams.",
          "data": "Video, deepfake detection from compressed-domain motion vectors",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Deepfake detection",
            "Detector robustness",
            "Temporal inconsistency cues"
          ],
          "key_terms": [
            "Deepfake detection",
            "Temporal inconsistency",
            "H.264 motion vectors",
            "Real-time detection",
            "FaceForensics++"
          ],
          "models": [
            "MTCNN",
            "MobileNetV3",
            "RAFT"
          ],
          "method_qualifiers": [
            "Two-stream network",
            "Compressed-domain motion features",
            "Data augmentation",
            "Cross-forgery generalisation"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "FaceForensics++: 1000 manipulated YouTube videos, five forgery types.",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness",
              "Temporal inconsistency cues"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness"
            ]
          },
          "tentative": false
        },
        {
          "title": "Digital Dilemmas: Humanitarian Consequences (exhibition)",
          "wid": "digital-dilemmas-humanitarian-consequences-exhibitio",
          "type": "event",
          "year": 2024,
          "venue": "EPFL Pavilions, 3 May to 14 July 2024. EPFL EssentialTech Centre with the ICRC, co-organised with EPFL Pavilions, in partnership with the Center for Digital Trust (C4DT)",
          "link": "https://epfl-pavilions.ch/fr/exhibitions/digital-dilemmas",
          "authors": [
            "Grégoire Castella (EssentialTech)",
            "Louis Potter (EssentialTech)",
            "Klaus Schönenberger (EssentialTech)",
            "Sarah Kenderdine (EPFL Pavilions)",
            "Marie Carrard (EPFL Pavilions)",
            "Peter Grönquist (IVRL)",
            "Stéphanie Milliquet (C4DT)",
            "Philippe Stoll (ICRC)",
            "Catherina Zazzini (ICRC)",
            "Fabrice Lauper"
          ],
          "epfl_authors": [
            "Grégoire Castella (EssentialTech)",
            "Louis Potter (EssentialTech)",
            "Klaus Schönenberger (EssentialTech)",
            "Sarah Kenderdine (EPFL Pavilions)",
            "Marie Carrard (EPFL Pavilions)",
            "Peter Grönquist (IVRL)",
            "Stéphanie Milliquet (C4DT)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_topics": [
            "Public exhibition",
            "Digital ethics",
            "Awareness",
            "ICRC"
          ],
          "stage": "Prevention",
          "relevance": 3,
          "about": "Immersive public exhibition at EPFL Pavilions exploring the digital risks that civilian populations and humanitarian workers face in conflict zones, where digital tools open access to essential services but can also expose personal data, enable surveillance or worsen misinformation. Visitors worked through dilemmas covering biometrics, civilian involvement in digital warfare, misinformation, AI-generated deepfakes, algorithmic decision-making, connectivity and data protection, and is followed by an installation of solutions developed by EPFL laboratories with the ICRC and ETH Zurich. It first appeared at UN headquarters in New York before this expanded Swiss presentation.",
          "why": "It addresses misinformation and AI deepfakes in humanitarian settings, raising public awareness.",
          "data": "not applicable (public exhibition)",
          "lab": [
            "EssentialTech Centre",
            "IVRL",
            "EPFL Pavilions",
            "C4DT",
            "ICRC"
          ],
          "themes": [
            "Media literacy & public resilience"
          ],
          "subtopics": [
            "Digital literacy",
            "Immersive demonstration",
            "Public awareness campaigns"
          ],
          "key_terms": [
            "Humanitarian action",
            "Deepfakes",
            "Hate speech",
            "Conflict zones",
            "Armed conflict"
          ],
          "models": [],
          "method_qualifiers": [],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Digital Dilemmas: Humanitarian Consequences",
              "kind": "platform"
            }
          ],
          "follow_up": [
            {
              "what": "Digital Dilemmas 2.0 - the \"Deepfake and You\" exhibit, built by the ICRC with EPFL's Image and Visual Representations Lab and shown in the foyer of the UN General Assembly building in New York",
              "kind": "successor-work",
              "url": "https://blogs.icrc.org/intercross/2025/03/20/digital-dilemmas-2-0-deepfakes/",
              "evidence": "ICRC Intercross podcast page, published 20 March 2025, titled \"Digital Dilemmas 2.0: Deepfakes\": \"We tour a deepfake exhibit created by the ICRC and L'Ecole Polytechnique Federale de Lausanne (EPFL) at the United Nations.\" In the transcript, EPFL research software engineer Peter Gronquist says: \"So we're at the UN right now. And I'm about to give you a short tour of this deep fake exhibit that we've been setting up with ICRC and EPFL.\" The page adds: \"The ICRC is working with EPFL to explain and find solutions to deepfakes and other technologies disseminating harmful information among populations in conflict areas.\""
            }
          ],
          "data_description": "No empirical data (exhibition description).",
          "follow_up_checked": "2026-09-02",
          "targeted_group": [
            "Civilian populations",
            "Civilians in conflict zones",
            "Humanitarian aid organisations",
            "Humanitarian aid workers"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Media literacy & public resilience": [
              "Public awareness campaigns",
              "Immersive demonstration",
              "Digital literacy"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media literacy & public resilience": [
              "Digital literacy",
              "Public awareness campaigns"
            ]
          },
          "tentative": true
        },
        {
          "title": "Leveraging Hierarchical Image-Text Misalignment for Universal Fake Image Detection",
          "wid": "leveraging-hierarchical-image-text-misalignment-for-uni",
          "type": "publication",
          "year": 2025,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2511.00427",
          "authors": [
            "Daichi Zhang",
            "Tong Zhang (EPFL)",
            "Jianmin Bao",
            "Shiming Ge",
            "Sabine Süsstrunk (IVRL, EPFL)"
          ],
          "epfl_authors": [
            "Tong Zhang (EPFL)",
            "Sabine Süsstrunk (IVRL, EPFL)"
          ],
          "about": "Detector for generated fake images that works from image-text misalignment rather than visual clues alone. The authors observe that fake images cannot be properly aligned with their captions the way real images can, and build a detector, ITEM, that measures this misalignment in pre-trained CLIP's joint visual-language space and tunes an MLP head on it. A hierarchical scheme looks first at the whole image and then at each semantic object named in the caption, so both global and fine-grained local misalignment act as clues. The motivation is that framing detection as naive binary image classification overfits specific image patterns and fails to generalise to unseen generative models.",
          "why": "Detection of AI-generated images, the production side of visual disinformation, with generalisation to generative models the detector has never seen.",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Cross-forgery generalisation",
            "Deepfake detection",
            "Detector robustness"
          ],
          "mdh_focus": [
            "D"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "Monitoring",
          "key_terms": [
            "AI-generated image detection",
            "Image-text misalignment",
            "Vision-language models",
            "Image forensics",
            "Deepfake detection"
          ],
          "models": [
            "BLIP-2",
            "CLIP",
            "GLIP",
            "ResNet-50",
            "Swin Transformer"
          ],
          "method_qualifiers": [
            "Cross-dataset generalisation",
            "Vision-language misalignment"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "ITEM",
              "kind": "model"
            }
          ],
          "data_description": "Images from 12 generators plus DiffusionForensics and GenImage benchmarks.",
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness",
              "Cross-forgery generalisation"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness"
            ]
          },
          "tentative": false
        },
        {
          "title": "Enhancing Frequency Forgery Clues for Diffusion-Generated Image Detection",
          "wid": "enhancing-frequency-forgery-clues-for-diffusion-generat",
          "type": "publication",
          "year": 2025,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2511.00429",
          "authors": [
            "Daichi Zhang",
            "Tong Zhang (EPFL)",
            "Shiming Ge",
            "Sabine Süsstrunk (IVRL, EPFL)"
          ],
          "epfl_authors": [
            "Tong Zhang (EPFL)",
            "Sabine Süsstrunk (IVRL, EPFL)"
          ],
          "about": "Detector for diffusion-generated images built on the observation that such images differ from natural real images by progressively larger amounts across low- to high-frequency bands. The method enhances a Frequency Forgery Clue across all frequency bands using a frequency-selective function that acts as a weighted filter on the Fourier spectrum, suppressing less discriminative bands and enhancing more informative ones. The stated aim is detection that generalises to unseen diffusion models and stays robust under various perturbations, which existing detectors struggle with.",
          "why": "Generalisable detection of diffusion-generated imagery, addressing concerns about malicious use of high-quality synthetic images.",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Cross-forgery generalisation",
            "Deepfake detection",
            "Detector robustness"
          ],
          "mdh_focus": [
            "D"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "Monitoring",
          "key_terms": [
            "Diffusion models",
            "Fourier spectrum",
            "Diffusion-generated image detection",
            "AI-generated image detection",
            "AIGC safety"
          ],
          "models": [
            "ADM",
            "ResNet-50",
            "Stable Diffusion"
          ],
          "method_qualifiers": [
            "Frequency domain analysis",
            "Cross-model generalization"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "40k real/fake images from GenImage and other public diffusion datasets.",
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Detector robustness",
              "Deepfake detection",
              "Cross-forgery generalisation"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "gatica",
      "name": "Daniel Gatica-Perez",
      "url": "https://people.epfl.ch/daniel.gatica-perez",
      "unit": "Idiap",
      "faculty": "IDIAP / IC",
      "mdh_focus": [
        "M",
        "D"
      ],
      "dataTypes": [
        "Text",
        "Video"
      ],
      "techTypes": [
        "NLP",
        "LLMs",
        "Media Framing Analysis",
        "Mixed Methods",
        "Computational Social Science",
        "Source Credibility"
      ],
      "stage": "Prevention + Monitoring",
      "publications": [
        {
          "title": "Framing Migration News with LLMs: Structured CoT as a Support for Human Interpretation",
          "wid": "framing-migration-news-with-llms-structured-cot-as-a-su",
          "type": "publication",
          "year": 2026,
          "venue": "COMPASS '26 (ACM SIGCAS/SIGCHI Conference on Computing and Sustainable Societies)",
          "link": "https://publications.idiap.ch/attachments/papers/2026/AlonsodelBarrio_ACMSIGCASSIGCHICONFERENCEONCOMPUTINGANDSUSTAINABLESOCIETIES_2026.pdf",
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Frame analysis",
            "Migration",
            "Local LLMs",
            "Structured chain-of-thought",
            "Interpretability",
            "Human-in-the-loop"
          ],
          "stage": "Monitoring",
          "relevance": 2,
          "about": "Study of whether a locally deployable open-source model can help researchers analyse how migration news is framed, used as an assistive tool rather than an automatic classifier. The method, Structured Chain-of-Thought prompting, makes the model give step-by-step justifications tied to predefined framing categories so a person can audit its reasoning. It outperforms few-shot and zero-shot baselines while running on a single GPU, and in a small human evaluation the explanations were generally seen as logical and prompted people to reconsider their first reading, though the authors caution that structured reasoning can also nudge human judgement.",
          "why": "An auditable frame-analysis tool for contested news that the authors place adjacent to misinformation detection.",
          "data": "Text, migration-related news articles",
          "themes": [
            "Media framing & narrative analysis"
          ],
          "subtopics": [
            "Frame identification",
            "Issue framing"
          ],
          "key_terms": [
            "Migration news",
            "Frame analysis",
            "Chain-of-thought prompting",
            "Human-in-the-loop annotation",
            "Interpretability"
          ],
          "models": [
            "Llama3-8B"
          ],
          "method_qualifiers": [
            "Few-shot prompting",
            "Local model deployment"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Structured Chain-of-Thought (SCoT) prompt for frame analysis",
              "kind": "framework"
            }
          ],
          "data_description": "700 migration news articles from Media Frames Corpus, 10 human annotators.",
          "follow_up_checked": "2026-09-02",
          "targeted_group": [
            "Migrants"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Media framing & narrative analysis": [
              "Issue framing",
              "Frame identification"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media framing & narrative analysis": [
              "Frame identification",
              "Issue framing"
            ]
          },
          "tentative": false
        },
        {
          "title": "Examining European Press Coverage of the Covid-19 No-Vax Movement: An NLP Framework",
          "wid": "examining-european-press-coverage-of-the-covid-19-no-va",
          "type": "publication",
          "year": 2023,
          "venue": "2nd ACM International Workshop on Multimedia AI against Disinformation (MAD '23), co-located with ACM ICMR",
          "link": "https://doi.org/10.1145/3592572.3592845",
          "authors": [
            "David Alonso del Barrio",
            "Daniel Gatica-Perez"
          ],
          "epfl_authors": [
            "David Alonso del Barrio",
            "Daniel Gatica-Perez"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "COVID-19",
            "Anti-vax",
            "European press",
            "NLP",
            "Sentiment analysis",
            "BERTopic"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "NLP study of how the European mainstream press covered the Covid-19 anti-vaccine movement and the disinformation tied to it. From a larger corpus it pulls 1786 no-vax articles from 19 newspapers across France, Italy, Spain, Switzerland, and the United Kingdom over 2020-2021, then runs subtopic modelling, country and political-orientation analysis, and a finer look at sentences mentioning disinformation, misinformation, hoaxes, rumours, or theories. About 21 percent of articles contain disinformation-related terms, coverage is negative across political orientations, and the most-named person in disinformation sentences is Bill Gates, followed by Trump and Biden.",
          "why": "Characterises how the European quality press countered anti-vax disinformation.",
          "data": "Text, 1786 articles from 19 newspapers across France, Italy, Spain, Switzerland, UK",
          "themes": [
            "Media framing & narrative analysis",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Coverage tone",
            "Human fact-checking",
            "Issue framing",
            "Outlet comparison",
            "Press debunking"
          ],
          "key_terms": [
            "No-vax movement",
            "European press",
            "5G conspiracy",
            "Bill Gates conspiracy theories",
            "Conspiracy theories"
          ],
          "models": [
            "BERTopic",
            "Spacy NER",
            "Word2Vec"
          ],
          "method_qualifiers": [
            "Topic modeling",
            "Sentiment analysis",
            "Named entity recognition"
          ],
          "events_cases": [
            "COVID-19 pandemic",
            "Covid-19 no-vax movement",
            "Covid-19 vaccination campaign"
          ],
          "built_at_epfl": [],
          "data_description": "1786 anti-vax articles from 19 European newspapers, 2020-2021.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "Europe",
            "France",
            "Italy",
            "Spain",
            "Switzerland",
            "United Kingdom"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Media framing & narrative analysis": [
              "Coverage tone",
              "Outlet comparison",
              "Issue framing"
            ],
            "Verification & content authenticity": [
              "Human fact-checking",
              "Press debunking"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media framing & narrative analysis": [
              "Coverage tone",
              "Issue framing",
              "Outlet comparison"
            ],
            "Verification & content authenticity": [
              "Human fact-checking"
            ]
          },
          "tentative": false
        },
        {
          "title": "How Did Europe's Press Cover Covid-19 Vaccination News? A Five-Country Analysis",
          "wid": "how-did-europe-s-press-cover-covid-19-vaccination-news-",
          "type": "publication",
          "year": 2022,
          "venue": "1st ACM International Workshop on Multimedia AI against Disinformation (MAD '22), co-located with ACM ICMR",
          "link": "https://doi.org/10.1145/3512732.3533588",
          "authors": [
            "David Alonso del Barrio",
            "Daniel Gatica-Perez"
          ],
          "epfl_authors": [
            "David Alonso del Barrio",
            "Daniel Gatica-Perez"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "COVID-19 vaccination",
            "European press",
            "NLP",
            "Sentiment analysis",
            "No-vax",
            "BERTopic"
          ],
          "stage": "Monitoring",
          "relevance": 2,
          "about": "Study of how the quality press framed Covid-19 vaccination across France, Italy, Spain, Switzerland and the UK over 22 months, using a corpus of 51320 articles. It applies named entity recognition, subtopic modelling and three-level sentiment analysis to break tone down by subtopic, government, country and vaccine brand. About 70 to 80 percent of coverage is neutral, consistent with an objectivity norm, while the no-vax subtopic is the most negatively toned everywhere and AstraZeneca draws the most negative sentiment.",
          "why": "A baseline characterisation of how the quality press presents vaccination news, mapping the ecosystem in which false claims circulate.",
          "data": "Text, 51320 Covid-19 vaccination articles from 19 newspapers across 5 countries",
          "themes": [
            "Media framing & narrative analysis"
          ],
          "subtopics": [
            "Coverage tone",
            "Outlet comparison"
          ],
          "key_terms": [
            "No-vax movement",
            "Covid-19 vaccination",
            "European press",
            "Topic modeling",
            "Infodemic"
          ],
          "models": [
            "BERTopic",
            "BERTsent",
            "Spacy NER"
          ],
          "method_qualifiers": [
            "Topic modeling",
            "Sentiment analysis",
            "Named entity recognition"
          ],
          "events_cases": [
            "COVID-19 pandemic",
            "Covid-19 vaccination"
          ],
          "built_at_epfl": [
            {
              "name": "European Covid-19 vaccination news dataset",
              "kind": "dataset",
              "evidence": "Paper full text: \"We constructed from scratch a dataset of European news articles about Covid-19 vaccination. As a first step, we contacted over 30 European newspapers spanning five countries, requesting authorization to extract and analyze articles discussing issues related to Covid-19 vaccination. We obtained the authorization of 19 of them.\" No public URL found: the paper has no data-availability statement and the authorisation obtained covers extraction and analysis, not redistribution."
            }
          ],
          "follow_up": [
            {
              "what": "Examining European Press Coverage of the Covid-19 No-Vax Movement: An NLP Framework (MAD '23, 2nd ACM International Workshop on Multimedia AI against Disinformation), same authors, built directly on this dataset",
              "kind": "successor-work",
              "url": "https://arxiv.org/abs/2305.00182",
              "evidence": "\"Our starting point has been a previous work (Alonso del Barrio and Gatica-Perez 2022), where we created a dataset of more than 50000 articles on Covid-19 vaccination with articles from Italy (2 newspapers), France (6 newspapers), Spain (6 newspapers), Switzerland (3 newspapers) and the United Kingdom (2 newspapers), with all the content translated to English.\" The reference list entry is: \"Alonso del Barrio, David and Daniel Gatica-Perez. 2022. How did Europe's press cover Covid-19 vaccination news? A five-country analysis. In Proceedings of the 1st International Workshop on Multimedia AI against Disinformation. 35-43.\""
            },
            {
              "what": "Framing the News: From Human Perception to Large Language Model Inferences (ACM ICMR 2023), same authors, framing analysis over a subset of the same corpus",
              "kind": "successor-work",
              "url": "https://arxiv.org/abs/2304.14456",
              "evidence": "\"We used part of the European Covid-19 News dataset collected in our recent work (Alonso del Barrio and Gatica-Perez 2022).\" and \"51320 articles on Covid-19 vaccination from 19 newspapers from 5 different countries: Italy, France, Spain, Switzerland and UK.\""
            }
          ],
          "data_description": "51320 articles from 19 newspapers, 5 countries, 22 months.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "France",
            "Italy",
            "Spain",
            "Switzerland",
            "United Kingdom"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Media framing & narrative analysis": [
              "Coverage tone",
              "Outlet comparison"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media framing & narrative analysis": [
              "Coverage tone",
              "Outlet comparison"
            ]
          },
          "tentative": false
        },
        {
          "title": "Referencing in YouTube Knowledge Communication Videos",
          "wid": "referencing-in-youtube-knowledge-communication-videos",
          "type": "publication",
          "year": 2023,
          "venue": "ACM International Conference on Interactive Media Experiences (IMX) 2023",
          "link": "https://doi.org/10.1145/3573381.3596163",
          "authors": [
            "Haeeun Kim",
            "Daniel Gatica-Perez"
          ],
          "epfl_authors": [
            "Haeeun Kim",
            "Daniel Gatica-Perez"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M"
          ],
          "mdh_topics": [
            "YouTube",
            "Science communication",
            "Citation",
            "Source credibility",
            "Digital literacy",
            "Video misinformation"
          ],
          "stage": "Prevention",
          "relevance": 4,
          "about": "Qualitative study of how YouTube creators of knowledge and educational videos cite their sources. The authors sample videos across channels, code them for referencing method, creator background and reference usability, then check whether the cited items are actually reachable. Most videos include a bibliography and many add in-video citations, but a large share of references turned out to be paywalled or impossible to retrieve. The authors propose platform design changes such as standardised reference fields and clickable in-player citations.",
          "why": "Treats transparent source-referencing on YouTube as a digital-literacy defence against video misinformation.",
          "data": "Video metadata, transcripts, descriptions, reference links",
          "themes": [
            "Media literacy & public resilience",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Digital literacy",
            "Reference accessibility",
            "Source referencing practices",
            "Trust indicators",
            "Voluntary gatekeeping"
          ],
          "key_terms": [
            "YouTube",
            "Science communication",
            "Information credibility",
            "Citation practices",
            "Digital literacy"
          ],
          "models": [],
          "method_qualifiers": [
            "Open coding",
            "Thematic analysis"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "44 English-speaking YouTube videos, 129 referenced resources, collected 2020-2022.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "YouTube"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Media literacy & public resilience": [
              "Digital literacy",
              "Voluntary gatekeeping"
            ],
            "Verification & content authenticity": [
              "Source referencing practices",
              "Trust indicators",
              "Reference accessibility"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media literacy & public resilience": [
              "Digital literacy",
              "Voluntary gatekeeping"
            ],
            "Verification & content authenticity": [
              "Source referencing practices",
              "Trust indicators"
            ]
          },
          "tentative": false
        },
        {
          "title": "Trust Indicators and Explainable AI: A Study on User Perceptions",
          "wid": "trust-indicators-and-explainable-ai-a-study-on-user-per",
          "type": "publication",
          "year": 2021,
          "venue": "INTERACT 2021 (IFIP TC13 Conference on Human-Computer Interaction), Springer LNCS 12934",
          "link": "https://doi.org/10.1007/978-3-030-85616-8_39",
          "authors": [
            "Delphine Ribes (EPFL+ECAL Lab)",
            "Nicolas Henchoz (EPFL+ECAL Lab)",
            "Hélène Portier (EPFL+ECAL Lab)",
            "Lara Defayes (EPFL+ECAL Lab)",
            "Thanh-Trung Phan (Idiap)",
            "Daniel Gatica-Perez (Idiap)",
            "Andreas Sonderegger (Bern UAS / Fribourg)"
          ],
          "epfl_authors": [
            "Delphine Ribes (EPFL+ECAL Lab)",
            "Nicolas Henchoz (EPFL+ECAL Lab)",
            "Hélène Portier (EPFL+ECAL Lab)",
            "Lara Defayes (EPFL+ECAL Lab)",
            "Thanh-Trung Phan (Idiap)",
            "Daniel Gatica-Perez (Idiap)"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M"
          ],
          "mdh_topics": [
            "Trust indicators",
            "Explainable AI",
            "News aggregators",
            "Interface design",
            "Source credibility"
          ],
          "stage": "Prevention",
          "relevance": 3,
          "about": "Experiment testing whether interface layout and explainable-AI cues change how much users trust an AI-driven news aggregator. Participants used a prototype that varied the explanation level and whether content was grouped by source type or ranked by a matching score, and the study measured trust, the perceived usefulness of citing sources, and felt versus actual understanding. Neither layout nor explanation level moved trust, but grouping by source raised how useful people found citing sources, and more detailed explanations worsened objective understanding while people still felt they understood. The authors argue the goal should be understandable AI rather than just explainable AI.",
          "why": "Tests how interface and explainable-AI cues shape trust and source-citation perception in a news aggregator.",
          "data": "Survey data, 226-participant online 2x3 between-subjects experiment",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Outlet credibility",
            "Source presentation",
            "Trust indicators"
          ],
          "key_terms": [
            "Fake news",
            "Trust indicators",
            "Explainable AI",
            "News aggregators",
            "Source credibility"
          ],
          "models": [],
          "method_qualifiers": [
            "Between-subjects experiment"
          ],
          "events_cases": [
            "Winegrowers' festival"
          ],
          "built_at_epfl": [],
          "data_description": "226 participants, online survey, 6 experimental conditions.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "Switzerland"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Trust indicators",
              "Outlet credibility",
              "Source presentation"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "Outlet credibility",
              "Trust indicators"
            ]
          },
          "tentative": false
        },
        {
          "title": "FAccT-Checked: A Narrative Review of Authority Reconfigurations and Retention in AI-Mediated Journalism",
          "wid": "facct-checked-a-narrative-review-of-authority-reconfigu",
          "type": "publication",
          "year": 2026,
          "venue": "arXiv preprint (manuscript submitted to ACM)",
          "link": "https://arxiv.org/abs/2604.21864",
          "authors": [
            "Stefano Sorrentino",
            "Matilde Barbini",
            "Daniel Gatica-Perez"
          ],
          "epfl_authors": [
            "Stefano Sorrentino",
            "Matilde Barbini",
            "Daniel Gatica-Perez"
          ],
          "about": "Critical narrative review across journalism studies, human-computer interaction and FAccT scholarship (final corpus of 209 records), conceptualizing editorial authority as the conjunction of decision rights, epistemic warrant and responsibility. It describes an internal migration, with editorial judgment progressively deferred to LLMs in newsroom workflows, and an external migration of decision-making power from news organizations toward platforms, vendors and infrastructural providers. Left unaddressed, these shifts risk making fairness hard to maintain, accountability difficult to assign and transparency performative. It also assesses the promise and structural limitations of participatory approaches as potential mechanisms for retaining or reclaiming editorial authority, and maps AI functionalities across journalistic workflows.",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Interpretability and oversight",
            "Misuse risk assessment",
            "Safety alignment"
          ],
          "mdh_focus": [
            "M"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "NA",
          "key_terms": [
            "Editorial authority",
            "AI-mediated journalism",
            "Platform dependency",
            "Participatory AI design",
            "FAccT"
          ],
          "models": [],
          "method_qualifiers": [
            "Narrative review",
            "Thematic analysis",
            "Conceptual framework"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "209 records from scholarly databases and grey literature, 1955-2025.",
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Safety alignment",
              "Interpretability and oversight",
              "Misuse risk assessment"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Interpretability and oversight",
              "Misuse risk assessment",
              "Safety alignment"
            ]
          },
          "tentative": false
        },
        {
          "title": "Generative AI Literacy: Twelve Defining Competencies",
          "wid": "generative-ai-literacy-twelve-defining-competencies",
          "type": "publication",
          "year": 2024,
          "venue": "Digital Government: Research and Practice, December 2024",
          "link": "https://arxiv.org/abs/2412.12107",
          "authors": [
            "Ravinithesh Annapureddy",
            "Alessandro Fornaroli",
            "Daniel Gatica-Perez"
          ],
          "epfl_authors": [
            "Ravinithesh Annapureddy",
            "Daniel Gatica-Perez"
          ],
          "about": "Framework defining generative AI literacy through twelve competencies, ranging from foundational AI literacy to prompt engineering and programming, and including ethical and legal considerations. It draws on a search of five scientific databases, which returned 6 records after removing duplicates, and a non-systematic literature review. The competencies include the ability to detect AI-generated content, whose absence may lead to misinformation that erodes trust and credibility, and the ability to assess outputs against trusted sources to prevent or minimize the effect of model hallucinations. The authors present the model as a roadmap for individuals getting familiar with generative AI and for researchers and policymakers developing assessments, educational programs, guidelines and regulations.",
          "themes": [
            "Media literacy & public resilience",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Critical thinking support",
            "Deepfake detection",
            "Detector robustness",
            "Digital literacy",
            "Hallucination fact-checking",
            "Legal awareness"
          ],
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "Prevention",
          "key_terms": [
            "Generative AI literacy",
            "AI-generated content detection",
            "Digital government",
            "Prompt engineering",
            "Competency model"
          ],
          "models": [],
          "method_qualifiers": [],
          "events_cases": [
            "Viral AI-generated Pentagon explosion image (2023)"
          ],
          "built_at_epfl": [
            {
              "name": "Generative AI Literacy competency-based model (12 competencies)",
              "kind": "framework"
            }
          ],
          "data_description": "No empirical data; conceptual literature-review paper.",
          "region_country": [
            "Switzerland"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Media literacy & public resilience": [
              "Digital literacy",
              "Critical thinking support",
              "Legal awareness"
            ],
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness",
              "Hallucination fact-checking"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media literacy & public resilience": [
              "Critical thinking support",
              "Digital literacy"
            ],
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness"
            ]
          },
          "tentative": false
        },
        {
          "title": "Human Interest or Conflict? Leveraging LLMs for Automated Framing Analysis in TV Shows",
          "wid": "human-interest-or-conflict-leveraging-llms-for-automate",
          "type": "publication",
          "year": 2024,
          "venue": "ACM IMX 2024",
          "link": "https://arxiv.org/abs/2409.12561",
          "authors": [
            "David Alonso del Barrio",
            "Max Tiel",
            "Daniel Gatica-Perez"
          ],
          "epfl_authors": [
            "Daniel Gatica-Perez"
          ],
          "about": "Study using prompt engineering with GPT-3.5 to identify the predominant frame in spoken content from Dutch television. The data are 2000 automatically transcribed items from the programs EenVandaag and Nieuwsuur, translated into English and labeled by a human annotator with five generic frames: human interest, conflict, economic, morality and attribution of responsibility. Agreement between annotator and model was 48.3 percent for EenVandaag and 38.7 percent for Nieuwsuur, and the model almost never matched the annotator's morality or responsibility labels. Framing experts and the annotator were interviewed about the disagreements, and the paper outlines uses such as support tools for journalists and educational resources for journalism students.",
          "themes": [
            "Media framing & narrative analysis"
          ],
          "subtopics": [
            "Frame identification",
            "Generic news frames",
            "TV news transcripts"
          ],
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "NA",
          "key_terms": [
            "Framing analysis",
            "Journalism support tools",
            "Prompt engineering",
            "Computational journalism",
            "Dutch public television"
          ],
          "models": [
            "GPT-3.5",
            "text-davinci-003"
          ],
          "method_qualifiers": [
            "Zero-shot prompting",
            "Manual annotation",
            "Expert elicitation"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "2000 Dutch TV news transcripts (2014-2018), translated to English",
          "platform": [
            "Television"
          ],
          "region_country": [
            "Netherlands"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Media framing & narrative analysis": [
              "Frame identification",
              "Generic news frames",
              "TV news transcripts"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media framing & narrative analysis": [
              "Frame identification"
            ]
          },
          "tentative": false
        },
        {
          "title": "Framing the News: From Human Perception to Large Language Model Inferences",
          "wid": "framing-the-news-from-human-perception-to-large-languag",
          "type": "publication",
          "year": 2023,
          "venue": "ACM ICMR 2023",
          "link": "https://infoscience.epfl.ch/server/api/core/bitstreams/27355684-dd82-4700-ba88-aecaf4a5cf9a/content",
          "authors": [
            "David Alonso del Barrio",
            "Daniel Gatica-Perez"
          ],
          "epfl_authors": [
            "Daniel Gatica-Perez"
          ],
          "about": "A study of news framing in 1786 headlines of articles about the Covid-19 no-vax movement from 19 newspapers in 5 European countries, with annotators labelling each headline's main frame using five generic frames plus a no-frame category. Human interest is the predominant frame (45.3 percent of headlines), followed by no frame (40.2 percent), with a relatively similar distribution between countries. Fine-tuned GPT-3.5 reaches 72 percent accuracy on the six-class task, 2 percentage points above RoBERTa, while prompt engineering with GPT-3.5 reaches 49 percent agreement with human labels. When annotators, without knowing the labels' origin, judged the model's labels on headlines where they had disagreed with it, agreement reached 76 percent.",
          "themes": [
            "Media framing & narrative analysis"
          ],
          "subtopics": [
            "Coverage tone",
            "Frame identification",
            "Outlet comparison"
          ],
          "mdh_focus": [
            "M"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 3,
          "stage": "NA",
          "key_terms": [
            "Covid-19 no-vax movement",
            "News framing",
            "Prompt engineering",
            "Vaccine hesitancy",
            "Computational journalism"
          ],
          "models": [
            "BERT",
            "GPT-3.5",
            "RoBERTa"
          ],
          "method_qualifiers": [
            "Zero-shot prompting",
            "Fine-tuning"
          ],
          "events_cases": [
            "Covid-19 anti-vaccine movement",
            "Covid-19 pandemic"
          ],
          "built_at_epfl": [],
          "data_description": "1786 headlines from 19 European newspapers, 2020-2021.",
          "region_country": [
            "France",
            "Italy",
            "Spain",
            "Switzerland",
            "United Kingdom"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Media framing & narrative analysis": [
              "Frame identification",
              "Outlet comparison",
              "Coverage tone"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media framing & narrative analysis": [
              "Coverage tone",
              "Frame identification",
              "Outlet comparison"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "kermarrec",
      "name": "Anne-Marie Kermarrec",
      "url": "https://people.epfl.ch/anne-marie.kermarrec",
      "unit": "SaCS",
      "faculty": "IC",
      "mdh_focus": [
        "M",
        "D",
        "H"
      ],
      "dataTypes": [
        "Data-agnostic",
        "Text"
      ],
      "techTypes": [
        "Network-based Detection",
        "Federated Learning",
        "Decentralized Learning",
        "Byzantine Robustness",
        "Privacy ML",
        "Fairness Auditing",
        "LLM Bias Audit",
        "ML Auditing"
      ],
      "stage": "Prevention + Monitoring + Mitigation",
      "publications": [
        {
          "title": "G-Fake: Tell Me How It Is Shared and I Shall Tell You If It Is Fake",
          "wid": "g-fake-tell-me-how-it-is-shared-and-i-shall-tell-you-if",
          "type": "publication",
          "year": 2022,
          "venue": "ACIIDS 2022 (Asian Conference on Intelligent Information and Database Systems), Springer CCIS 1716",
          "link": "https://doi.org/10.1007/978-981-19-8234-7_1",
          "authors": [
            "Nawfal Abbassi Saber (UM6P)",
            "Rachid Guerraoui (EPFL)",
            "Anne-Marie Kermarrec (EPFL)",
            "Alexandre Maurer (UM6P)"
          ],
          "epfl_authors": [
            "Rachid Guerraoui (EPFL)",
            "Anne-Marie Kermarrec (EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "misinformation:detection",
            "platform-moderation:fake-news",
            "network-based-detection",
            "Fake news detection",
            "Graph embedding",
            "Network-based detection",
            "Influence graph",
            "Early detection"
          ],
          "stage": "Mitigation + Monitoring",
          "relevance": 5,
          "about": "Fake-news detector that works only from how a story is shared, with no access to its text, images or the social graph itself. It builds an influence graph linking users who co-shared items, embeds each user into a vector, and classifies an item from the average embedding of its first sharers. Using only the first 30 sharing actions per item it reaches 96.8 percent accuracy, beating a text-based classifier trained on news titles, and reconstructs the influence graph far faster than a prior baseline.",
          "why": "A content-free fake-news classifier driven by who shares a story rather than what it says.",
          "data": "Graph/Network (FakeNewsNet)",
          "dataTypes": [
            "Data-agnostic"
          ],
          "techTypes": [
            "Network-based Detection",
            "Node Embedding",
            "Graph Signal Processing"
          ],
          "themes": [
            "Spread, amplification & networks",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Account coordination graphs",
            "Claim credibility inference",
            "Detector robustness",
            "Early-stage detection",
            "Information cascades"
          ],
          "key_terms": [
            "Fake news detection",
            "Influence graph",
            "Graph embedding",
            "Content-agnostic detection",
            "Early detection"
          ],
          "models": [
            "BERT",
            "Node2Vec"
          ],
          "method_qualifiers": [
            "Graph embedding",
            "Network inference",
            "Unsupervised representation learning"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "G-Fake",
              "kind": "framework"
            }
          ],
          "data_description": "FakeNewsNet: 21595 labeled news items, 1.7M tweets, 535367 users.",
          "platform": [
            "Twitter/X"
          ],
          "region_country": [
            "United States"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Spread, amplification & networks": [
              "Information cascades",
              "Account coordination graphs"
            ],
            "Verification & content authenticity": [
              "Claim credibility inference",
              "Detector robustness",
              "Early-stage detection"
            ]
          },
          "theme_qualifiers_canonical": {
            "Spread, amplification & networks": [
              "Account coordination graphs",
              "Information cascades"
            ],
            "Verification & content authenticity": [
              "Claim credibility inference",
              "Detector robustness"
            ]
          },
          "tentative": false
        },
        {
          "title": "The Fake News Vaccine: A Content-Agnostic System for Preventing Fake News from Becoming Viral",
          "wid": "the-fake-news-vaccine",
          "type": "publication",
          "year": 2019,
          "venue": "NETYS 2019 (International Conference on Networked Systems), Springer LNCS 11704",
          "link": "https://doi.org/10.1007/978-3-030-31277-0_23",
          "authors": [
            "Oana Balmau (University of Sydney)",
            "Rachid Guerraoui (EPFL)",
            "Anne-Marie Kermarrec (Inria)",
            "Alexandre Maurer (EPFL)",
            "Matej Pavlovic (EPFL)",
            "Willy Zwaenepoel (University of Sydney)"
          ],
          "epfl_authors": [
            "Rachid Guerraoui (EPFL)",
            "Anne-Marie Kermarrec (Inria)",
            "Alexandre Maurer (EPFL)",
            "Matej Pavlovic (EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "misinformation:prevention",
            "platform-moderation:viral-content",
            "human-in-the-loop",
            "Fake news prevention",
            "Content-agnostic detection",
            "Crowdsourced fact-checking",
            "Viral suppression",
            "Twitter-scale evaluation"
          ],
          "stage": "Monitoring + Mitigation",
          "relevance": 5,
          "about": "A platform-side plugin, Credulix, that stops fake-news items from going viral without reading the content of posts. Human fact-checkers review only the most viral items, and the system records how each user reacts to build a per-user credulity profile, then uses a Bayesian procedure to estimate the probability that an unchecked item is fake given who has shared it. Once that probability crosses a high threshold the item is removed from other feeds. Tested on a large Twitter graph, it correctly flagged over 99 percent of unchecked fake items after fact-checkers reviewed only 1024, with low overhead.",
          "why": "A platform-side system that prevents fake news from going viral while reviewing only a fraction of items.",
          "data": "Graph/Network (41M-user Twitter graph, 35M+ tweet corpus)",
          "dataTypes": [
            "Data-agnostic"
          ],
          "techTypes": [
            "Bayesian Inference",
            "Human-in-the-Loop",
            "Platform Plugin"
          ],
          "themes": [
            "Content moderation & enforcement",
            "Spread, amplification & networks",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Claim credibility inference",
            "Moderator-assist detection",
            "News-feed suppression",
            "Plugin-based moderation",
            "Spread interventions",
            "Viral prevention"
          ],
          "key_terms": [
            "Fake news",
            "Content-agnostic detection",
            "Virality prevention",
            "User credulity records",
            "Bayesian inference"
          ],
          "models": [],
          "method_qualifiers": [
            "Bayesian inference",
            "Naive Bayes",
            "Incremental computation",
            "Simulation-based evaluation"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Credulix",
              "kind": "tool",
              "evidence": "No URL found. The paper describes the artefact as built and evaluated ('We implement Credulix as a standalone Java plugin and connect it to Twissandra [14] (an open source Twitter clone), which serves as a baseline system.') but never releases or links it. The only GitHub link in the whole paper is to the third-party baseline it plugs into: 'Twissandra Twitter clone, build on top of cassandra. https://github.com/twissandra/twissandra/'."
            }
          ],
          "data_description": "41M-user Twitter graph, 35M tweets, simulated fake news propagation.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "News-feed suppression",
              "Moderator-assist detection",
              "Plugin-based moderation"
            ],
            "Spread, amplification & networks": [
              "Spread interventions",
              "Viral prevention"
            ],
            "Verification & content authenticity": [
              "Claim credibility inference"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Moderator-assist detection",
              "News-feed suppression"
            ],
            "Spread, amplification & networks": [
              "Information cascades",
              "Spread interventions"
            ],
            "Verification & content authenticity": [
              "Claim credibility inference",
              "Human fact-checking"
            ]
          },
          "tentative": false
        },
        {
          "title": "The Universal Gossip Fighter",
          "wid": "the-universal-gossip-fighter",
          "type": "publication",
          "year": 2022,
          "venue": "IEEE IPDPS 2022 (International Parallel and Distributed Processing Symposium)",
          "link": "https://doi.org/10.1109/IPDPS53621.2022.00116",
          "authors": [
            "Anastasiia Gorbunova",
            "Rachid Guerraoui",
            "Anne-Marie Kermarrec",
            "Anastasiia Kucherenko",
            "Rafael Pinot"
          ],
          "epfl_authors": [
            "Anastasiia Gorbunova",
            "Rachid Guerraoui",
            "Anne-Marie Kermarrec",
            "Anastasiia Kucherenko",
            "Rafael Pinot"
          ],
          "mdh_relevance": "infrastructure",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "misinformation:adversarial-robustness",
            "distributed-systems:byzantine",
            "Gossip protocols",
            "Distributed computing",
            "Misinformation spread",
            "Network adversaries",
            "Fake news spread"
          ],
          "stage": "NA",
          "relevance": 2,
          "about": "A theoretical adversary built to slow down any all-to-all gossip protocol in a partially synchronous system, without knowing in advance which protocol it faces. With a limited budget it crashes, isolates, or delays processes, using randomisation so the protocol cannot detect and adapt to the attack. The paper proves it forces either much higher time or much higher message complexity on any such protocol, and experiments turn logarithmic-time spreading into linear time. One stated motivation is modelling how the spread of viral fake news or epidemic messages might be hampered.",
          "why": "It models how the propagation of viral fake-news messages could be slowed at the protocol level.",
          "data": "Graph/Network, simulated distributed systems",
          "dataTypes": [
            "Data-agnostic"
          ],
          "techTypes": [
            "Adversarial Distributed Computing",
            "Byzantine Robustness",
            "Gossip Protocols"
          ],
          "themes": [
            "NA"
          ],
          "subtopics": [],
          "key_terms": [
            "Gossip protocols",
            "Adaptive adversary",
            "Fake news dissemination",
            "Distributed computing",
            "Communication complexity"
          ],
          "models": [],
          "method_qualifiers": [
            "Adaptive adversary",
            "Lower-bound analysis",
            "Randomized adversarial strategy"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Universal Gossip Fighter",
              "kind": "tool",
              "url": "https://gitlab.epfl.ch/kucheren/the-universal-gossip-fighter",
              "evidence": "Paper Section V: \"For reproducibility purposes, our implementation is accessible online\" with footnote 1 giving the EPFL GitLab URL. The project page (opened) is public and reads: name \"The Universal Gossip Fighter\", owner Anastasiia Kucherenko, description \"The framework that evaluates the impact of the Universal Gossip Fighter on the time and message complexities of several gossip protocols in various system configurations.\""
            }
          ],
          "data_description": "Simulated all-to-all gossip protocols with up to 500 processes.",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {},
          "theme_qualifiers_canonical": {},
          "tentative": false
        },
        {
          "title": "LLM Demographic Bias Audit (early-stage)",
          "wid": "llm-demographic-bias-audit-early-stage",
          "type": "project",
          "year": "2025-ongoing",
          "venue": "EPFL SaCS",
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "D",
            "H"
          ],
          "mdh_topics": [
            "ai-safety:bias",
            "llm:audit",
            "ai-impact:detection"
          ],
          "stage": "Monitoring",
          "relevance": 2,
          "data": "Text, matched LLM prompts varied on demographic attributes",
          "dataTypes": [
            "Text"
          ],
          "techTypes": [
            "LLM Auditing",
            "Bias Detection"
          ],
          "themes": [
            "NA"
          ],
          "subtopics": [
            "Demographic bias probing",
            "Identity-conditioned responses"
          ],
          "key_terms": [
            "Inferred user identity",
            "LLM demographic bias",
            "LLM-mediated moderation",
            "Model auditing"
          ],
          "models": [],
          "method_qualifiers": [
            "Algorithmic auditing"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "Matched LLM prompts varied on demographic attributes; audit started 2025.",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Demographic bias probing",
              "Identity-conditioned responses"
            ]
          },
          "theme_qualifiers_canonical": {},
          "tentative": true
        },
        {
          "title": "The Evaluation Game: Beyond Static LLM Benchmarking",
          "wid": "the-evaluation-game-beyond-static-llm-benchmarking",
          "type": "publication",
          "year": 2026,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2605.19377",
          "authors": [
            "Paul Wang",
            "Jade Garcia Bourrée",
            "Anne-Marie Kermarrec",
            "Vincent Corruble"
          ],
          "epfl_authors": [
            "Jade Garcia Bourrée",
            "Anne-Marie Kermarrec"
          ],
          "about": "Study modelling robustness fine-tuning against jailbreaks as a two-player game between an evaluator who audits a model for jailbreaks and a trainer who fine-tunes it, with data augmentation represented as group actions. On the circle with cyclic translation groups, below a critical threshold of the trainer's generalization range the evaluator maintains a constant miss ratio for linearly many rounds. Experiments with Llama, Qwen and Mistral on WildJailBreak prompts give significant evidence that fine-tuning on adversarial prompts induces only local generalization, with refusal rates on test examples highly correlated with their distance to the fine-tuning prompts. The authors conclude that static benchmarks cannot distinguish a genuine fix from a memorized patch.",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Adaptive attacks",
            "Adversarial robustness",
            "Safety alignment"
          ],
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "Prevention + Monitoring",
          "key_terms": [
            "Robustness fine-tuning",
            "Adversarial evaluation",
            "AI safety",
            "EU AI Act",
            "Game theory"
          ],
          "models": [
            "GPT-2",
            "Llama-3.1-8B-Instruct",
            "Mistral-7B-Instruct-v0.3",
            "Pythia-410m",
            "Qwen2.5-7B-Instruct",
            "gpt2"
          ],
          "method_qualifiers": [
            "Jailbreaking",
            "LoRA",
            "Group actions",
            "Adversarial training"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "WildJailbreak prompts: 47 training, 2999 held-out; seven open LLMs.",
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Adaptive attacks",
              "Adversarial robustness",
              "Safety alignment"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Adaptive attacks",
              "Adversarial robustness",
              "Safety alignment"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "salzmann",
      "name": "Mathieu Salzmann",
      "url": "https://people.epfl.ch/mathieu.salzmann",
      "unit": "CVLab",
      "faculty": "IC / SDSC",
      "mdh_focus": [
        "M",
        "D",
        "H"
      ],
      "dataTypes": [
        "Text",
        "Image"
      ],
      "techTypes": [
        "Content Filtering",
        "Computer Vision",
        "Training Data Attribution",
        "ML Privacy",
        "Adversarial ML"
      ],
      "stage": "Prevention + Monitoring",
      "publications": [
        {
          "title": "LoRIF: Low-Rank Influence Functions for Scalable Training Data Attribution",
          "wid": "lorif-low-rank-influence-functions-for-scalable-trainin",
          "type": "publication",
          "year": 2026,
          "venue": "arXiv (Li, Le, Xu, Salzmann)",
          "link": "https://arxiv.org/abs/2601.21929",
          "authors": [
            "Shuangqi Li (CVLab)",
            "Hieu Le (CVLab)",
            "Jingyi Xu (Stony Brook)",
            "Mathieu Salzmann (CVLab)"
          ],
          "epfl_authors": [
            "Shuangqi Li (CVLab)",
            "Hieu Le (CVLab)",
            "Mathieu Salzmann (CVLab)"
          ],
          "mdh_relevance": "infrastructure",
          "mdh_focus": [
            "H"
          ],
          "mdh_topics": [
            "Training data attribution",
            "Influence functions",
            "Safety auditing",
            "Data poisoning",
            "Apertus-70B"
          ],
          "stage": "NA",
          "relevance": 2,
          "about": "LoRIF performs training data attribution, identifying which training examples most influenced a given model output, useful for debugging, data curation, safety auditing, and spotting data contamination or poisoning. It makes gradient-based influence functions scale to large language models by exploiting low-rank structure in the gradients, storing compact factors instead of full matrices. On models from 0.1B to 70B parameters it matches or exceeds a prior baseline's attribution quality at up to 20x lower gradient storage and query time, and it was evaluated on Apertus-70B. It is not itself a misinformation or hate-speech detector.",
          "why": "The same machinery lets a model owner trace a generated output back to the training examples responsible, the basis for safety auditing and data-poisoning diagnosis.",
          "data": "Text, model gradients (GPT2-small on WikiText-103, OLMo-3-7B, Apertus-70B)",
          "themes": [
            "NA"
          ],
          "subtopics": [],
          "key_terms": [
            "Training data attribution",
            "Influence functions",
            "Safety auditing",
            "Low-rank approximation",
            "LLM auditing"
          ],
          "models": [
            "Apertus-70B",
            "GPT2",
            "GPT2-small",
            "OLMo-3-7B"
          ],
          "method_qualifiers": [
            "LLM-as-a-judge",
            "Influence functions"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "LoRIF",
              "kind": "framework"
            }
          ],
          "follow_up": [
            {
              "what": "LoRIF reference implementation released on GitHub (official PyTorch code for the paper, with the end-to-end GPT-2 / WikiText-103 workflow at the paper's low-storage configuration f=16, c=1, r=2048)",
              "kind": "repository",
              "url": "https://github.com/doub7e/LoRIF",
              "evidence": "Repository description: \"Official implementation for the paper \\\"LoRIF: Low-Rank Influence Functions for Scalable Training Data Attribution\\\"\". README: \"# LoRIF\n\nOfficial PyTorch implementation of **[LoRIF: Low-Rank Influence Functions for Scalable Training Data Attribution](https://arxiv.org/abs/2601.21929)**.\""
            }
          ],
          "data_description": "Models from 124M to 70B parameters, 233K to 3.8M attribution examples.",
          "follow_up_checked": "2026-09-02",
          "targeted_group": [
            "LGBTQ+"
          ],
          "label_version": "v5",
          "theme_qualifiers": {},
          "theme_qualifiers_canonical": {},
          "tentative": false
        },
        {
          "title": "Apertus Multimodal Content Filtering",
          "wid": "apertus-multimodal-content-filtering",
          "type": "project",
          "year": "2025-ongoing",
          "venue": "SDSC, within the Swiss AI Initiative (EPFL + ETH Zurich + CSCS)",
          "link": "https://www.apertus-ai.org/",
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_topics": [
            "Apertus",
            "Multimodal content filtering",
            "Swiss AI Initiative",
            "Training data curation",
            "LLM safety",
            "Open-source LLM",
            "Data compliance",
            "Multilingual",
            "Digital sovereignty"
          ],
          "stage": "NA",
          "relevance": 4,
          "about": "In-progress data-curation effort for the next, multimodal phase of Apertus, Switzerland's open-source large language model. Salzmann's group works on the image side, helping filter training content so the corpus is clean before training, while other labs lead the broader multimodal phase. There are no published results yet because the work is at the data-pipeline stage; the September 2025 public Apertus release is text-only.",
          "why": "Filtering what goes into a widely used national model shapes the misinformation, disinformation, and hateful content it can later produce.",
          "data": "Image + Text, training corpora for Apertus",
          "themes": [
            "NA"
          ],
          "subtopics": [
            "Training-corpus design"
          ],
          "key_terms": [
            "Apertus",
            "Swiss AI Initiative",
            "Training-data curation",
            "LLM safety",
            "Content filtering"
          ],
          "models": [
            "Apertus"
          ],
          "method_qualifiers": [
            "Training-data filtering"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "follow_up": [
            {
              "what": "Apertus, the open multilingual model this data-curation work feeds. Listed separately on this map as an ongoing project under Bosselut.",
              "kind": "related-work",
              "url": "https://www.apertus-ai.org/",
              "evidence": "The entry's own description: \"In-progress data-curation effort for the next, multimodal phase of Apertus, Switzerland's open-source large language model.\" No separate page describes the multimodal filtering effort itself."
            },
            {
              "what": "Apertus technical report, \"Apertus: Democratizing Open and Compliant LLMs for Global Language Environments\", accepted at ACL 2026. Listed separately on this map as a publication under Bosselut.",
              "kind": "related-work",
              "url": "https://arxiv.org/abs/2509.14233",
              "evidence": "The report documents the released text-only Apertus, including its toxicity filtering of pretraining data. It predates the multimodal phase this entry covers and does not describe it."
            }
          ],
          "data_description": "Multimodal training corpora for Apertus LLM.",
          "follow_up_checked": "2026-09-09",
          "region_country": [
            "Switzerland"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Training-corpus design"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Training-corpus design"
            ]
          },
          "tentative": true
        }
      ]
    },
    {
      "id": "bugnion",
      "name": "Edouard Bugnion",
      "url": "https://people.epfl.ch/edouard.bugnion",
      "unit": "Vice Presidency for Innovation and Impact (VPI)",
      "faculty": "IC",
      "mdh_focus": [
        "M",
        "D"
      ],
      "dataTypes": [
        "Text",
        "Image"
      ],
      "techTypes": [
        "Internet Measurement",
        "Platform Analysis",
        "Observatory Model",
        "Innovation Strategy"
      ],
      "stage": "Monitoring + Mitigation",
      "publications": []
    },
    {
      "id": "dillenbourg",
      "name": "Pierre Dillenbourg",
      "url": "https://people.epfl.ch/pierre.dillenbourg",
      "unit": "CHILI Lab",
      "faculty": "IC (emeritus)",
      "mdh_focus": [],
      "dataTypes": [],
      "techTypes": [
        "Strategy"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "cavallaro",
      "name": "Andrea Cavallaro",
      "url": "https://people.epfl.ch/andrea.cavallaro",
      "unit": "MINTS",
      "faculty": "STI",
      "mdh_focus": [
        "H",
        "M",
        "D"
      ],
      "dataTypes": [
        "Video",
        "Text",
        "Audio",
        "Image"
      ],
      "techTypes": [
        "Multimodal Fusion",
        "Hate Speech Detection",
        "LLM Embeddings",
        "Privacy Engineering",
        "AI Alignment",
        "Adversarial ML",
        "AI Ethics Teaching"
      ],
      "stage": "Prevention + Monitoring",
      "publications": [
        {
          "title": "MM-HSD: Multi-Modal Hate Speech Detection in Videos",
          "wid": "mm-hsd-multi-modal-hate-speech-detection-in-videos",
          "type": "publication",
          "year": 2025,
          "venue": "ACM Multimedia 2025 (MM '25)",
          "link": "https://doi.org/10.1145/3746027.3754558",
          "authors": [
            "Berta Céspedes-Sarrias",
            "Carlos Collado-Capell",
            "Pablo Rodenas-Ruiz",
            "Olena Hrynenko",
            "Andrea Cavallaro"
          ],
          "epfl_authors": [
            "Berta Céspedes-Sarrias",
            "Carlos Collado-Capell",
            "Pablo Rodenas-Ruiz",
            "Olena Hrynenko",
            "Andrea Cavallaro"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "H"
          ],
          "mdh_topics": [
            "hate-speech:detection",
            "hate-speech:multimodal",
            "hate-speech:video"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "Hate-speech detector for videos that combines four signals at once: the spoken transcript, the raw audio, the video frames, and any on-screen text. It runs cross-modal attention over the raw embeddings as an early feature extractor, then concatenates that output with modality-specific encoders before a binary classifier, and sweeps which modality serves as query versus keys. On the HateMM benchmark it reaches macro-F1 0.874, beating the prior state of the art (HCC1 at 0.848) and the original HateMM baseline (0.790), and it targets the case where image, soundtrack, and overlaid text are each harmless alone and only become hateful in combination.",
          "why": "It detects hate speech in videos, including the case where each modality is neutral on its own.",
          "data": "Video, HateMM dataset (1083 BitChute videos, 39.8% hate)",
          "dataTypes": [
            "Video",
            "Audio",
            "Text",
            "Image",
            "Multimodal"
          ],
          "techTypes": [
            "Cross-Modal Attention",
            "Multimodal Fusion",
            "ViT",
            "wav2vec2",
            "Hate Speech Detection"
          ],
          "themes": [
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Multi-modal hate cues",
            "On-screen text cues",
            "Video hate speech"
          ],
          "key_terms": [
            "Hate speech detection",
            "Cross-modal attention",
            "Multi-modal fusion",
            "On-screen text",
            "BitChute"
          ],
          "models": [
            "Detoxify",
            "MM-HSD",
            "PaddleOCR",
            "RoBERTa",
            "ViT",
            "Whisper",
            "wav2vec2"
          ],
          "method_qualifiers": [
            "Cross-modal attention",
            "Optical character recognition",
            "Feature fusion"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "MM-HSD",
              "kind": "model",
              "url": "https://github.com/idiap/mm-hsd",
              "evidence": "Paper, footnote 1 to \"We release MM-HSD as an open-source benchmark for video-based HSD to support and advance ongoing research in this area 1\": \"1 https://github.com/idiap/mm-hsd\". Repository README: \"This is the code accompanying the paper 'MM-HSD: Multi-Modal Hate Speech Detection in Videos' by B. Cespedes-Sarrias, C. Collado-Capell, P. Rodenas-Ruiz, O. Hrynenko, and A. Cavallaro, published in the Proceedings of the 33rd ACM International Conference on Multimedia (ACM MM '25).\""
            }
          ],
          "data_description": "1083 labeled videos from BitChute, 43 hours.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "BitChute"
          ],
          "targeted_group": [
            "Ethnic minorities",
            "LGBTQ+",
            "Religious minorities"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Toxicity & harassment": [
              "Multi-modal hate cues",
              "On-screen text cues",
              "Video hate speech"
            ]
          },
          "theme_qualifiers_canonical": {
            "Toxicity & harassment": [
              "Multi-modal hate cues"
            ]
          },
          "tentative": false
        },
        {
          "title": "Specializing General-purpose LLM Embeddings for Implicit Hate Speech Detection across Datasets",
          "wid": "specializing-general-purpose-llm-embeddings-for-implici",
          "type": "publication",
          "year": 2025,
          "venue": "2nd International Workshop on Diffusion of Harmful Content on Online Web (DHOW '25), co-located with ACM MM 2025",
          "link": "https://doi.org/10.1145/3746275.3762209",
          "authors": [
            "Vassiliy Cheremetiev",
            "Quang Long Ho Ngo",
            "Chau Ying Kot",
            "Andrea Cavallaro"
          ],
          "epfl_authors": [
            "Vassiliy Cheremetiev",
            "Quang Long Ho Ngo",
            "Chau Ying Kot",
            "Andrea Cavallaro"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "H"
          ],
          "mdh_topics": [
            "hate-speech:detection",
            "hate-speech:implicit",
            "llm:specialization"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "Study that fine-tunes off-the-shelf LLM embedding models to detect implicit hate speech, meaning prejudice expressed through indirect language, sarcasm, or coded terms rather than open slurs. The fine-tuning needs no external knowledge or extra context data and is tested both within a single dataset and across datasets. Within a dataset the gains are small (up to 1.10 F1-macro points), suggesting general-purpose embeddings already sit near the task-specific ceiling when train and test data match, while across datasets specialization adds up to 20.35 F1-macro points, so the main payoff is generalizing across different annotation schemes.",
          "why": "It targets implicit hate speech expressed through indirect language, sarcasm, or coded terms.",
          "data": "Text, implicit hate speech datasets (Latent Hatred, ToxiGen, SBIC)",
          "dataTypes": [
            "Text"
          ],
          "techTypes": [
            "LLM Embeddings",
            "Contrastive Fine-Tuning",
            "Hate Speech Detection"
          ],
          "themes": [
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Cross-dataset evaluation",
            "Detector target bias",
            "Implicit hate speech"
          ],
          "key_terms": [
            "Implicit hate speech",
            "Content moderation",
            "Cross-dataset generalization",
            "LLM embeddings",
            "Hate speech detection"
          ],
          "models": [
            "BERT",
            "BERTweet",
            "E5",
            "Gemma-7B",
            "Jasper",
            "Llama2",
            "Llama3-8B",
            "NV-Embed",
            "Qwen",
            "Qwen3-8B",
            "RoBERTa",
            "Stella"
          ],
          "method_qualifiers": [
            "LoRA",
            "Linear probing",
            "Embedding fine-tuning",
            "Cross-dataset generalisation"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "implicit-hsd",
              "kind": "tool"
            }
          ],
          "follow_up": [
            {
              "what": "Code release for this paper: idiap/implicit-hsd on GitHub (Apache-2.0). Fine-tuning and evaluation code for the LLM-embedding IHS classifiers on Implicit Hate Corpus, SBIC, DynaHate and ToxiGen.",
              "kind": "repository",
              "url": "https://github.com/idiap/implicit-hsd",
              "evidence": "Paper sec. 1, verbatim: \"The code is available at https://github.com/idiap/implicit-hsd.\" The repository's own description is the paper title, \"Specializing General-purpose LLM Embeddings for Implicit Hate Speech Detection across Datasets\", and its citation block reads: \"Cheremetiev, Vassiliy and Ngo, Quang Long Ho and Kot, Chau Ying and Baia, Alina Elena and Cavallaro, Andrea. 2025. Specializing General-purpose LLM Embeddings for Implicit Hate Speech Detection across Datasets. In Proceedings of the 2nd International Workshop on Diffusion of Harmful Content on Online Web (DHOW '25).\" with DOI https://doi.org/10.1145/3746275.3762209, matching this work exactly."
            }
          ],
          "data_description": "4 IHS datasets (IHC, DynaHate, SBIC, ToxiGen), ~114k social media posts.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Reddit",
            "Twitter/X"
          ],
          "region_country": [
            "United States"
          ],
          "targeted_group": [
            "Black people",
            "Immigrants",
            "Jewish people",
            "LGBTQ+",
            "Muslims",
            "Racial and ethnic minorities",
            "Religious minorities",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Toxicity & harassment": [
              "Implicit hate speech",
              "Detector target bias",
              "Cross-dataset evaluation"
            ]
          },
          "theme_qualifiers_canonical": {
            "Toxicity & harassment": [
              "Detector target bias",
              "Implicit hate speech"
            ]
          },
          "tentative": false
        },
        {
          "title": "EE-559 Deep Learning, group project themed \"Deep learning to foster safer online spaces\"",
          "wid": "ee-559-deep-learning-group-project-themed-deep-learning",
          "type": "project",
          "year": "2022-ongoing",
          "venue": "EPFL Deep Learning course (EE-559)",
          "link": "https://memento.epfl.ch/event/deep-learning-students-tackle-online-hate-speech-2/",
          "authors": [
            "Andrea Cavallaro"
          ],
          "epfl_authors": [
            "Andrea Cavallaro"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "H"
          ],
          "mdh_topics": [
            "hate-speech:education",
            "ai-ethics:teaching",
            "ai-ethics:dual-use"
          ],
          "stage": "NA",
          "relevance": 4,
          "about": "EPFL deep learning course whose group project is built around hate-speech detection, so students build such systems as hands-on work. The course covers modern deep learning along with the ethics of data acquisition and model deployment, plus interpretability, bias, and fairness. By the end of May 2026, about 550 students will have built hate-speech systems through the course, and a public showcase has roughly 150 students present their projects. The hate-speech framing asks students who decides what counts as hate speech versus sarcasm and confronts them with labelling ambiguity and the asymmetry between false positives and false negatives.",
          "why": "Its group project trains hundreds of students per year to build hate-speech detectors and reason about the choices that involves.",
          "data": "not applicable (course)",
          "dataTypes": [
            "Text"
          ],
          "techTypes": [
            "Hate Speech Detection",
            "AI Ethics Teaching"
          ],
          "themes": [
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Implicit hate speech",
            "Multi-modal hate cues",
            "Satire disambiguation"
          ],
          "key_terms": [
            "Hate speech detection",
            "Deep learning coursework",
            "Student poster session",
            "Deep learning",
            "Deep learning course"
          ],
          "models": [],
          "method_qualifiers": [],
          "events_cases": [],
          "built_at_epfl": [],
          "follow_up": [
            {
              "what": "EE-559 Deep Learning coursebook entry, the formal course description. It documents the course itself; it does not mention the hate-speech project theme.",
              "kind": "course-page",
              "url": "https://edu.epfl.ch/coursebook/en/deep-learning-EE-559",
              "evidence": "Coursebook summary: the course \"explores how to design reliable discriminative and generative neural networks, the ethics of data acquisition and model deployment, as well as modern multi-modal models.\" Assessment: \"Exercises and group project\". No mention of hate speech, misinformation or moderation appears anywhere on that page."
            }
          ],
          "data_description": "Event page for a student poster session; no empirical dataset described.",
          "follow_up_checked": "2026-09-10",
          "label_version": "v5",
          "theme_qualifiers": {
            "Toxicity & harassment": [
              "Multi-modal hate cues",
              "Satire disambiguation",
              "Implicit hate speech"
            ]
          },
          "theme_qualifiers_canonical": {
            "Toxicity & harassment": [
              "Implicit hate speech",
              "Multi-modal hate cues"
            ]
          },
          "tentative": true
        }
      ]
    },
    {
      "id": "al-hassanieh",
      "name": "Haitham Al-Hassanieh",
      "url": "https://people.epfl.ch/haitham.alhassanieh",
      "unit": "SENS Lab",
      "faculty": "IC",
      "mdh_focus": [
        "M",
        "D",
        "H"
      ],
      "dataTypes": [],
      "techTypes": [
        "AI Ethics Teaching",
        "Humanitarian AI",
        "Research Funding Review"
      ],
      "stage": "Prevention",
      "publications": []
    },
    {
      "id": "nesti",
      "name": "Alessandro Nesti",
      "url": "https://people.epfl.ch/alessandro.nesti",
      "unit": "SDSC",
      "faculty": "Lead, Digital Society Vertical (SDSC)",
      "mdh_focus": [
        "D",
        "H"
      ],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "NLP",
        "Information Extraction",
        "AI Literacy",
        "Humanitarian AI",
        "NGO Partnerships"
      ],
      "stage": "Prevention + Monitoring",
      "publications": [
        {
          "title": "Geneva AI Initiative, Humanitarian AI Literacy",
          "wid": "geneva-ai-initiative-humanitarian-ai-literacy",
          "type": "project",
          "year": "2024-ongoing",
          "venue": "SDSC + EPFL AI Center",
          "link": "https://ai.epfl.ch/innovation/geneva-ai-initiative/",
          "mdh_relevance": "infrastructure",
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_topics": [
            "Geneva AI Initiative",
            "Humanitarian AI literacy",
            "ICAIN",
            "Capacity building",
            "EPFL AI Center"
          ],
          "stage": "NA",
          "relevance": 2,
          "about": "Capacity-building programme that runs AI-literacy training for humanitarian organisations and then helps them frame projects and write funding proposals. The first edition ran without a training component and most resulting projects failed to produce adoptable solutions because the organisations were not ready, so the current version invests heavily in upskilling first. It feeds the operational humanitarian projects, including the MDH-relevant humanitarian proposals, with ICAIN providing a parallel funding channel.",
          "why": "It feeds the MDH-relevant humanitarian projects, so its link to disinformation runs through the pipeline rather than direct.",
          "data": "not applicable (initiative)",
          "lab": [
            "SDSC",
            "EPFL AI Center"
          ],
          "themes": [
            "NA"
          ],
          "subtopics": [],
          "key_terms": [
            "AI literacy",
            "Capacity building",
            "ICAIN",
            "Geneva AI Initiative",
            "Humanitarian organisations"
          ],
          "models": [],
          "method_qualifiers": [],
          "events_cases": [],
          "built_at_epfl": [],
          "follow_up": [
            {
              "what": "DASHI, a four-day introductory AI course for humanitarian and international-organisation professionals, run as the first step of the initiative",
              "kind": "deployment",
              "url": "https://ai.epfl.ch/innovation/geneva-ai-initiative/",
              "evidence": "An accessible introduction to AI concepts, strategy, and practical use cases — designed for professionals in the humanitarian and international sectors."
            }
          ],
          "data_description": "Initiative description; no empirical data.",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {},
          "theme_qualifiers_canonical": {},
          "tentative": true
        }
      ]
    },
    {
      "id": "stadler",
      "name": "Theresa Stadler",
      "url": "https://people.epfl.ch/theresa.stadler",
      "unit": "SPRING Lab / SDSC",
      "faculty": "IC",
      "mdh_focus": [
        "D",
        "H"
      ],
      "dataTypes": [
        "Text",
        "Data-agnostic",
        "Image"
      ],
      "techTypes": [
        "Privacy Engineering",
        "Synthetic Data",
        "Differential Privacy",
        "OSINT",
        "Adversarial Stress-Testing",
        "Policy"
      ],
      "stage": "Prevention",
      "publications": []
    },
    {
      "id": "guerraoui",
      "name": "Rachid Guerraoui",
      "url": "https://people.epfl.ch/rachid.guerraoui",
      "unit": "DCL",
      "faculty": "IC",
      "dataTypes": [
        "Text",
        "Data-agnostic"
      ],
      "techTypes": [
        "Federated Learning",
        "Fairness",
        "Misinformation Research"
      ],
      "stage": "Monitoring + Mitigation",
      "publications": [
        {
          "title": "G-Fake: Tell Me How It is Shared and I Shall Tell You If It is Fake",
          "wid": "g-fake-tell-me-how-it-is-shared-and-i-shall-tell-you-if",
          "type": "publication",
          "year": 2022,
          "venue": "ACIIDS 2022 (Asian Conference on Intelligent Information and Database Systems), Springer CCIS 1716",
          "link": "https://doi.org/10.1007/978-981-19-8234-7_1",
          "authors": [
            "Nawfal Abbassi Saber (UM6P)",
            "Rachid Guerraoui (DCL, EPFL)",
            "Anne-Marie Kermarrec (SaCS, EPFL)",
            "Alexandre Maurer (UM6P)"
          ],
          "epfl_authors": [
            "Rachid Guerraoui (DCL, EPFL)",
            "Anne-Marie Kermarrec (SaCS, EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "misinformation:detection",
            "platform-moderation:fake-news",
            "network-based-detection",
            "Fake news detection",
            "Graph embedding",
            "Network-based detection",
            "Influence graph",
            "Early detection"
          ],
          "stage": "Mitigation + Monitoring",
          "relevance": 5,
          "about": "Fake-news detector that works purely from how a story spreads, with no access to its text, images, or the underlying social graph. It uses only the ordered history of who shared each item: it rebuilds an influence graph linking users who co-shared a story, turns each user into an embedding, and a classifier labels a story fake or real from the embeddings of its first sharers. Using only the first 30 shares it reaches 96.8 percent accuracy, beating a content-based classifier trained on news titles.",
          "why": "A content-agnostic classifier that flags fake news from sharing patterns alone.",
          "data": "Graph/Network (FakeNewsNet, 21595 items, 1.7M tweets, 535367 users)",
          "themes": [
            "Spread, amplification & networks",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Account coordination graphs",
            "Claim credibility inference",
            "Detector robustness",
            "Early-stage detection",
            "Information cascades"
          ],
          "key_terms": [
            "Fake news detection",
            "Influence graph",
            "Graph embedding",
            "Content-agnostic detection",
            "Early detection"
          ],
          "models": [
            "BERT",
            "Node2Vec"
          ],
          "method_qualifiers": [
            "Graph embedding",
            "Network inference",
            "Unsupervised representation learning"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "G-Fake",
              "kind": "framework"
            }
          ],
          "data_description": "FakeNewsNet: 21595 labeled news items, 1.7M tweets, 535367 users.",
          "platform": [
            "Twitter/X"
          ],
          "region_country": [
            "United States"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Spread, amplification & networks": [
              "Information cascades",
              "Account coordination graphs"
            ],
            "Verification & content authenticity": [
              "Claim credibility inference",
              "Detector robustness",
              "Early-stage detection"
            ]
          },
          "theme_qualifiers_canonical": {
            "Spread, amplification & networks": [
              "Account coordination graphs",
              "Information cascades"
            ],
            "Verification & content authenticity": [
              "Claim credibility inference",
              "Detector robustness"
            ]
          },
          "tentative": false
        },
        {
          "title": "The Fake News Vaccine: A Content-Agnostic System for Preventing Fake News from Becoming Viral",
          "wid": "the-fake-news-vaccine",
          "type": "publication",
          "year": 2019,
          "venue": "NETYS 2019 (International Conference on Networked Systems), Springer LNCS 11704",
          "link": "https://doi.org/10.1007/978-3-030-31277-0_23",
          "authors": [
            "Oana Balmau (University of Sydney)",
            "Rachid Guerraoui (EPFL)",
            "Anne-Marie Kermarrec (Inria)",
            "Alexandre Maurer (EPFL)",
            "Matej Pavlovic (EPFL)",
            "Willy Zwaenepoel (University of Sydney)"
          ],
          "epfl_authors": [
            "Rachid Guerraoui (EPFL)",
            "Anne-Marie Kermarrec (Inria)",
            "Alexandre Maurer (EPFL)",
            "Matej Pavlovic (EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "misinformation:prevention",
            "platform-moderation:viral-content",
            "human-in-the-loop",
            "Fake news prevention",
            "Content-agnostic detection",
            "Crowdsourced fact-checking",
            "Viral suppression",
            "Twitter-scale evaluation"
          ],
          "stage": "Monitoring + Mitigation",
          "relevance": 5,
          "about": "A platform-side plugin, Credulix, that stops fake-news items from going viral without reading the content of posts. Human fact-checkers review only the most viral items, and the system records how each user reacts to build a per-user credulity profile, then uses a Bayesian procedure to estimate the probability that an unchecked item is fake given who has shared it. Once that probability crosses a high threshold the item is removed from other feeds. Tested on a large Twitter graph, it correctly flagged over 99 percent of unchecked fake items after fact-checkers reviewed only 1024, with low overhead.",
          "why": "A platform-side system that prevents fake news from going viral while reviewing only a fraction of items.",
          "data": "Graph/Network (41M-user Twitter graph, 35M+ tweet corpus)",
          "themes": [
            "Content moderation & enforcement",
            "Spread, amplification & networks",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Claim credibility inference",
            "Moderator-assist detection",
            "News-feed suppression",
            "Plugin-based moderation",
            "Spread interventions",
            "Viral prevention"
          ],
          "key_terms": [
            "Fake news",
            "Content-agnostic detection",
            "Virality prevention",
            "User credulity records",
            "Bayesian inference"
          ],
          "models": [],
          "method_qualifiers": [
            "Bayesian inference",
            "Naive Bayes",
            "Incremental computation",
            "Simulation-based evaluation"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Credulix",
              "kind": "tool",
              "evidence": "No URL found. The paper describes the artefact as built and evaluated ('We implement Credulix as a standalone Java plugin and connect it to Twissandra [14] (an open source Twitter clone), which serves as a baseline system.') but never releases or links it. The only GitHub link in the whole paper is to the third-party baseline it plugs into: 'Twissandra Twitter clone, build on top of cassandra. https://github.com/twissandra/twissandra/'."
            }
          ],
          "data_description": "41M-user Twitter graph, 35M tweets, simulated fake news propagation.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "News-feed suppression",
              "Moderator-assist detection",
              "Plugin-based moderation"
            ],
            "Spread, amplification & networks": [
              "Spread interventions",
              "Viral prevention"
            ],
            "Verification & content authenticity": [
              "Claim credibility inference"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Moderator-assist detection",
              "News-feed suppression"
            ],
            "Spread, amplification & networks": [
              "Information cascades",
              "Spread interventions"
            ],
            "Verification & content authenticity": [
              "Claim credibility inference",
              "Human fact-checking"
            ]
          },
          "tentative": false
        },
        {
          "title": "The Universal Gossip Fighter",
          "wid": "the-universal-gossip-fighter",
          "type": "publication",
          "year": 2022,
          "venue": "IEEE IPDPS 2022 (International Parallel and Distributed Processing Symposium)",
          "link": "https://doi.org/10.1109/IPDPS53621.2022.00116",
          "authors": [
            "Anastasiia Gorbunova",
            "Rachid Guerraoui",
            "Anne-Marie Kermarrec",
            "Anastasiia Kucherenko",
            "Rafael Pinot"
          ],
          "epfl_authors": [
            "Anastasiia Gorbunova",
            "Rachid Guerraoui",
            "Anne-Marie Kermarrec",
            "Anastasiia Kucherenko",
            "Rafael Pinot"
          ],
          "mdh_relevance": "infrastructure",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "misinformation:adversarial-robustness",
            "distributed-systems:byzantine",
            "Gossip protocols",
            "Distributed computing",
            "Misinformation spread",
            "Network adversaries",
            "Fake news spread"
          ],
          "stage": "NA",
          "relevance": 2,
          "about": "A theoretical adversary built to slow down any all-to-all gossip protocol in a partially synchronous system, without knowing in advance which protocol it faces. With a limited budget it crashes, isolates, or delays processes, using randomisation so the protocol cannot detect and adapt to the attack. The paper proves it forces either much higher time or much higher message complexity on any such protocol, and experiments turn logarithmic-time spreading into linear time. One stated motivation is modelling how the spread of viral fake news or epidemic messages might be hampered.",
          "why": "It models how the propagation of viral fake-news messages could be slowed at the protocol level.",
          "data": "Simulated distributed systems (N from 10 to 500, F from 0.1N to 0.5N)",
          "themes": [
            "NA"
          ],
          "subtopics": [],
          "key_terms": [
            "Gossip protocols",
            "Adaptive adversary",
            "Fake news dissemination",
            "Distributed computing",
            "Communication complexity"
          ],
          "models": [],
          "method_qualifiers": [
            "Adaptive adversary",
            "Lower-bound analysis",
            "Randomized adversarial strategy"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Universal Gossip Fighter",
              "kind": "tool",
              "url": "https://gitlab.epfl.ch/kucheren/the-universal-gossip-fighter",
              "evidence": "Paper Section V: \"For reproducibility purposes, our implementation is accessible online\" with footnote 1 giving the EPFL GitLab URL. The project page (opened) is public and reads: name \"The Universal Gossip Fighter\", owner Anastasiia Kucherenko, description \"The framework that evaluates the impact of the Universal Gossip Fighter on the time and message complexities of several gossip protocols in various system configurations.\""
            }
          ],
          "data_description": "Simulated all-to-all gossip protocols with up to 500 processes.",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {},
          "theme_qualifiers_canonical": {},
          "tentative": false
        },
        {
          "title": "Stochastic Parrots Looking for Stochastic Parrots: LLMs are Easy to Fine-Tune and Hard to Detect with other LLMs",
          "wid": "stochastic-parrots-looking-for-stochastic-parrots-llms-",
          "type": "publication",
          "year": 2023,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2304.08968",
          "authors": [
            "Da Silva Gameiro Henrique",
            "Andrei Kucharavy",
            "Rachid Guerraoui"
          ],
          "epfl_authors": [
            "Da Silva Gameiro Henrique",
            "Andrei Kucharavy",
            "Rachid Guerraoui"
          ],
          "about": "Study of how well detectors of LLM-generated text can be countered, since LLMs can write articles pushing disinformation or supporting harassment campaigns. With GPT-2 small as generator and BERT base as discriminator, fine-tuning GPT-2 on the detector's human reference texts left BERT unable to distinguish its samples even with all outputs correctly labeled, while training GPT-2 on BERT's scores either failed to evade it or evaded it but degenerated. Separately, pairing reinforcement from a critic model with the AdamW optimizer rather than Adam allowed well-generalizing fine-tunes from limited data. The authors see the results as strongly arguing against continued use of LLMs fine-tuned for classification to detect generative LLMs in the wild.",
          "themes": [
            "AI safety",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "AI-generated text detection",
            "Detector robustness",
            "Misuse risk assessment",
            "Safety alignment"
          ],
          "mdh_focus": [
            "D"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 3,
          "stage": "Prevention + Monitoring",
          "key_terms": [
            "Reinforcement from critic",
            "Text GANs",
            "Adversarial evasion",
            "AdamW fine-tuning",
            "BERT"
          ],
          "models": [
            "BERT",
            "DPGAN",
            "GPT-2"
          ],
          "method_qualifiers": [
            "Adversarial evasion",
            "Reinforcement from critic"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "10K MS COCO and 280K EMNLP news text excerpts for adversarial training.",
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Misuse risk assessment",
              "Safety alignment",
              "Adversarial robustness"
            ],
            "Verification & content authenticity": [
              "Detector robustness",
              "AI-generated text detection"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Adversarial robustness",
              "Misuse risk assessment",
              "Safety alignment"
            ],
            "Verification & content authenticity": [
              "AI-generated text detection",
              "Detector robustness"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "frossard",
      "name": "Pascal Frossard",
      "url": "https://people.epfl.ch/pascal.frossard",
      "unit": "LTS4",
      "faculty": "STI",
      "mdh_focus": [
        "D"
      ],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "Adversarial ML",
        "Robustness",
        "LLM Poisoning",
        "Explainable AI",
        "Deep Learning"
      ],
      "stage": "Prevention",
      "publications": [
        {
          "title": "NMT-Obfuscator Attack: Ignore a sentence in translation with only one word",
          "wid": "nmt-obfuscator-attack-ignore-a-sentence-in-translation-",
          "type": "publication",
          "year": 2024,
          "venue": "NeurIPS 2024 Safe Generative AI Workshop",
          "link": "https://arxiv.org/abs/2411.12473",
          "authors": [
            "Sahar Sadrizadeh",
            "César Descalzo",
            "Ljiljana Dolamic",
            "Pascal Frossard"
          ],
          "epfl_authors": [
            "Sahar Sadrizadeh",
            "César Descalzo",
            "Pascal Frossard"
          ],
          "about": "Paper proposing NMT-Obfuscator, an adversarial attack on neural machine translation models that finds one word to add between two sentences so that the second sentence is not translated, while the whole text stays natural in the source language. The authors note that an attacker can use it to hide malicious information in automatic translation. Tested on WMT14 English-to-French and English-to-German translation with Marian NMT, and English-to-French with mBART50, the attack keeps the second sentence out of the translation in 76, 50 and 71 percent of cases on average, respectively, a success rate competitive with the Suffix-Dropper baseline but with much lower perplexity.",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Adaptive attacks",
            "Adversarial robustness",
            "Misuse risk assessment"
          ],
          "mdh_focus": [
            "D"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "NA",
          "key_terms": [
            "Neural machine translation",
            "Adversarial attack",
            "Adversarial Attack",
            "Adversarial attacks",
            "Covert communication"
          ],
          "models": [
            "DeepL",
            "GPT-2",
            "Marian NMT",
            "mBART50"
          ],
          "method_qualifiers": [
            "Adversarial evasion",
            "White-box attack",
            "Gradient projection"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "NMT-Obfuscator",
              "kind": "tool"
            }
          ],
          "data_description": "WMT14 En-Fr/En-De test sets, 3003 sentences each.",
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Adaptive attacks",
              "Adversarial robustness",
              "Misuse risk assessment"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Adaptive attacks",
              "Adversarial robustness",
              "Misuse risk assessment"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "salathe",
      "name": "Marcel Salathé",
      "url": "https://people.epfl.ch/marcel.salathe",
      "unit": "Digital Epidemiology Lab",
      "faculty": "SV",
      "mdh_focus": [
        "M",
        "D"
      ],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "NLP",
        "LLMs",
        "Digital Epidemiology",
        "Stance Classification",
        "Sentiment Analysis",
        "AI Literacy"
      ],
      "stage": "Prevention + Monitoring + Mitigation",
      "publications": [
        {
          "title": "Addressing Machine Learning Concept Drift Reveals Declining Vaccine Sentiment During the COVID-19 Pandemic",
          "wid": "addressing-machine-learning-concept-drift-reveals-decli",
          "type": "publication",
          "year": 2020,
          "venue": "arXiv 2012.02197 (Digital Epidemiology Lab, EPFL)",
          "link": "https://arxiv.org/abs/2012.02197",
          "authors": [
            "Martin Müller",
            "Marcel Salathé"
          ],
          "epfl_authors": [
            "Marcel Salathé"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M"
          ],
          "mdh_topics": [
            "Vaccine sentiment",
            "Concept drift",
            "Misclassification",
            "COVID-19",
            "Twitter",
            "Stance classification"
          ],
          "stage": "Monitoring",
          "relevance": 2,
          "about": "Study of how stance classifiers for vaccine tweets degrade over time, a problem called concept drift where a model trained on older data does worse on newer data. The authors annotated a subset of vaccine-related tweets for positive, neutral, or negative stance and compared a frozen 2018 model against models retrained periodically. They find the anti-vaccine class drifts fastest, with model performance declining over 20 percent across three years and most of that drop concentrated in about ten months in early 2020. A model left unchanged from 2018 would have largely missed the real decline in vaccine sentiment during 2020.",
          "why": "It shows uncorrected model drift systematically misclassifies anti-vaccine content and misses real swings in health sentiment.",
          "data": "Text, 57.5M English vaccination-related tweets (Jul 1 2017 to Oct 1 2020) from 9.9M unique users; 11893 three-fold annotated tweets",
          "themes": [
            "Persuasion & cognitive effects"
          ],
          "subtopics": [],
          "key_terms": [
            "Concept drift",
            "Vaccine sentiment",
            "COVID-19",
            "Twitter",
            "Anti-vaccine content"
          ],
          "models": [
            "BERT",
            "FastText"
          ],
          "method_qualifiers": [
            "Concept drift evaluation",
            "Sliding-window retraining",
            "Sentiment analysis"
          ],
          "events_cases": [
            "COVID-19 pandemic"
          ],
          "built_at_epfl": [
            {
              "name": "concept_drift_paper",
              "kind": "dataset"
            }
          ],
          "follow_up": [
            {
              "what": "Public code and data repository for the paper (digitalepidemiologylab/concept_drift_paper): annotation files, drift experiment scripts and figure-generation code",
              "kind": "repository",
              "url": "https://github.com/digitalepidemiologylab/concept_drift_paper",
              "evidence": "Paper, Data availability section: \"All data and code can be found on our public GitHub repository https://github.com/ digitalepidemiologylab/concept_drift_paper .\" The repository page confirms: \"This repository contains data & code necessary to reproduce the paper\"."
            },
            {
              "what": "Zenodo release of the tweet IDs used in the study: \"English Vaccine-related Twitter data (Tweet IDs) 2018-07-01 to 2020-09-30\", by Martin Muller and Marcel Salathe (EPFL), published 29 November 2020, CC BY 4.0",
              "kind": "dataset",
              "url": "https://doi.org/10.5281/zenodo.4295829",
              "evidence": "Repository README: \"# Tweet IDs\nAll tweet IDs which were used for the study are available on Zenodo:\" followed by the badge for DOI 10.5281/zenodo.4295829. The Zenodo record itself is titled \"English Vaccine-related Twitter data (Tweet IDs) 2018-07-01 to 2020-09-30\", creators Martin Muller (EPFL) and Marcel Salathe (EPFL), and describes tweets \"collected via the Crowdbreaks platform\" matching vaccination-related keywords."
            }
          ],
          "data_description": "57.5M English vaccine tweets, 2017-2020; 11893 annotated for stance.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "label_version": "v5",
          "theme_qualifiers": {},
          "theme_qualifiers_canonical": {},
          "tentative": false
        },
        {
          "title": "The Dynamics of Health Behavior Sentiments on a Large Online Social Network",
          "wid": "the-dynamics-of-health-behavior-sentiments-on-a-large-o",
          "type": "publication",
          "year": 2013,
          "venue": "EPJ Data Science 2:4",
          "link": "https://doi.org/10.1140/epjds16",
          "authors": [
            "Marcel Salathé",
            "Duy Q. Vu",
            "Shashank Khandelwal",
            "David R. Hunter"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M"
          ],
          "mdh_topics": [
            "Vaccine sentiment",
            "Social contagion",
            "Homophily",
            "Health behavior",
            "Twitter",
            "Network analysis"
          ],
          "stage": "Monitoring",
          "relevance": 2,
          "about": "Early study of how opinions about vaccination spread between people on Twitter, using time-stamped tweets about intent to get the pandemic H1N1 vaccine, classified as positive, neutral, or negative. The authors fit survival models separating how the size of a user's opinionated neighborhood, mutual ties, and exposure intensity relate to what the user later expresses. Exposure to negative sentiment predicts more negative expression afterward, consistent with social contagion, while exposure to positive sentiment does not produce a matching positive effect. The measured dynamics favor the spread of negative vaccination sentiment but not positive.",
          "why": "It shows negative vaccination sentiment spreads asymmetrically online, a foundational dynamic behind health misinformation.",
          "data": "Text + follower/followee graph; Twitter H1N1 vaccination sentiment tweets (positive/neutral/negative), final estimates based on the last 25 days of collection",
          "themes": [
            "Persuasion & cognitive effects",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Exposure intensity",
            "Homophily",
            "Information cascades",
            "Peer influence",
            "Vaccine hesitancy"
          ],
          "key_terms": [
            "Vaccine hesitancy",
            "Social contagion",
            "Homophily",
            "H1N1 vaccine",
            "Twitter"
          ],
          "models": [],
          "method_qualifiers": [
            "Cox proportional hazards"
          ],
          "events_cases": [
            "H1N1 influenza pandemic"
          ],
          "built_at_epfl": [],
          "data_description": "477768 tweets from over 100000 Twitter users, 4.8M follower edges.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "region_country": [
            "United States"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Persuasion & cognitive effects": [
              "Peer influence",
              "Vaccine hesitancy"
            ],
            "Spread, amplification & networks": [
              "Exposure intensity",
              "Information cascades",
              "Homophily"
            ]
          },
          "theme_qualifiers_canonical": {
            "Persuasion & cognitive effects": [
              "Peer influence",
              "Vaccine hesitancy"
            ],
            "Spread, amplification & networks": [
              "Echo chambers",
              "Exposure intensity",
              "Information cascades"
            ]
          },
          "tentative": false
        },
        {
          "title": "Using Large Language Models to Understand the Public Discourse Towards Vaccination in Brazil Between January 2013 and December 2019",
          "wid": "using-large-language-models-to-understand-the-public-di",
          "type": "publication",
          "year": 2025,
          "venue": "medRxiv 2025.02.24.25322766 (Digital Epidemiology Lab, EPFL)",
          "link": "https://doi.org/10.1101/2025.02.24.25322766",
          "authors": [
            "Laura Espinosa",
            "Marcel Salathé"
          ],
          "epfl_authors": [
            "Laura Espinosa",
            "Marcel Salathé"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M"
          ],
          "mdh_topics": [
            "Vaccine stance",
            "Brazil",
            "LLM classification",
            "Measles resurgence",
            "Twitter",
            "Public health monitoring"
          ],
          "stage": "Monitoring",
          "relevance": 3,
          "about": "Study tracking how public stance toward vaccination shifted in Brazil from 2013 to 2019 using Portuguese-language vaccine tweets, classifying stance with an LLM and building a vaccine stance index compared against vaccination coverage and confirmed measles cases. Of about 1.7 million classified tweets, negative stance rose over the period and peaked in 2019, coinciding with the 2018-2019 measles resurgence, and increased measles cases were predominantly followed by lower vaccine stance in the same or following years. Negative tweets often carried incorrect information about vaccine safety and distrust of health and national authorities, and Brazil lost its measles-free status in 2019.",
          "why": "It links LLM-measured online vaccine stance, including tweets carrying incorrect safety information, to a real measles resurgence.",
          "data": "Text, 2197090 Portuguese vaccine-related tweets from Brazil (Jan 1 2013 to Dec 31 2019); 1703009 classified by GPT-4",
          "themes": [
            "Persuasion & cognitive effects"
          ],
          "subtopics": [
            "Attitude change",
            "Public stance",
            "Vaccine hesitancy"
          ],
          "key_terms": [
            "Vaccine hesitancy",
            "Measles resurgence",
            "Brazil",
            "Twitter",
            "Measles outbreak"
          ],
          "models": [
            "GPT-4",
            "GPT-4-o-mini",
            "GPT-4o-mini"
          ],
          "method_qualifiers": [
            "Few-shot prompting",
            "Temporal trend analysis"
          ],
          "events_cases": [
            "Brazil measles outbreak 2018-2019",
            "Brazil yellow fever outbreak 2018",
            "Measles resurgence in Brazil 2018-2019",
            "Yellow fever epidemic"
          ],
          "built_at_epfl": [
            {
              "name": "brazil-vaccine-uptake",
              "kind": "dataset"
            }
          ],
          "follow_up": [
            {
              "what": "brazil-vaccine-uptake repository (Digital Epidemiology Lab): the R and Python code plus tweet identifiers and summarised anonymised datasets for this study",
              "kind": "repository",
              "url": "https://github.com/digitalepidemiologylab/brazil-vaccine-uptake",
              "evidence": "Data sharing section of the paper: \"The data (list of tweets identifiers and summarised anonymised datasets) and R and Python code used can be found in the online repository at https://github.com/digitalepidemiologylab/brazil-vaccine-uptake.\" The methods section also states: \"The code and list of R packages used are available in the repository 'brazil-vaccine-uptake'.\" The repository page itself reads \"Prediction of vaccine uptake in Brazil based on Twitter vaccine sentiment\" and its README says it contains \"code and data to collect and analyse the stance towards vaccination in Brazil from X (former Twitter) data\", noting that full non-anonymised Twitter data is withheld \"for data privacy\" with tweet IDs given instead, which matches the paper's statement exactly."
            }
          ],
          "data_description": "2.2M Portuguese tweets from Brazil, 2013-2019.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "region_country": [
            "Brazil"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Persuasion & cognitive effects": [
              "Vaccine hesitancy",
              "Attitude change",
              "Public stance"
            ]
          },
          "theme_qualifiers_canonical": {
            "Persuasion & cognitive effects": [
              "Attitude change",
              "Vaccine hesitancy"
            ]
          },
          "tentative": false
        },
        {
          "title": "Use of Large Language Models as a Scalable Approach to Understanding Public Health Discourse",
          "wid": "use-of-large-language-models-as-a-scalable-approach-to-",
          "type": "publication",
          "year": 2024,
          "venue": "PLOS Digital Health 3(10):e0000631",
          "link": "https://doi.org/10.1371/journal.pdig.0000631",
          "authors": [
            "Laura Espinosa",
            "Marcel Salathé"
          ],
          "epfl_authors": [
            "Laura Espinosa",
            "Marcel Salathé"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M"
          ],
          "mdh_topics": [
            "Stance detection",
            "Vaccine sentiment",
            "LLM",
            "GPT-4",
            "Mixtral",
            "Crowdsourcing"
          ],
          "stage": "Monitoring",
          "relevance": 2,
          "about": "Methodological study of how well large language models can read public stance toward vaccination from social media. The authors compared crowdsourcing, a rule-based baseline, and several LLMs against a gold standard set by experts. Few-shot prompting of the best models came out on top and beat crowdsourcing, while poorly engineered zero-shot prompting sometimes did worse than the rule-based baseline. They note that all current methods carry a real risk of substantial misclassification, and that an open model that can run locally matched the best commercial one, useful when data cannot be sent to an external API.",
          "why": "It establishes which LLMs reliably classify vaccine stance, the methodological backbone for scaling up health-misinformation monitoring.",
          "data": "Text, 1000 random English-language vaccination tweets (Dec 2 2019 to Mar 11 2022); expert, crowd, LLM and VADER annotations",
          "themes": [
            "Persuasion & cognitive effects"
          ],
          "subtopics": [
            "Public health surveillance",
            "Stance detection",
            "Vaccine hesitancy"
          ],
          "key_terms": [
            "Public health surveillance",
            "Stance detection",
            "Vaccine hesitancy",
            "Social listening",
            "Crowd versus LLM annotation"
          ],
          "models": [
            "GPT-3.5",
            "GPT-4",
            "Llama 3",
            "Llama3 (70B)",
            "Mistral (7B)",
            "Mixtral (8x7B)"
          ],
          "method_qualifiers": [
            "Few-shot prompting",
            "Stance detection"
          ],
          "events_cases": [
            "COVID-19 pandemic"
          ],
          "built_at_epfl": [
            {
              "name": "llm_crowd_experts_annotation",
              "kind": "tool"
            }
          ],
          "follow_up": [
            {
              "what": "Code and data release for the study: tweet identifiers, summarised anonymised datasets, and the R and Python classification scripts, under Salathe's EPFL Digital Epidemiology Lab GitHub organisation",
              "kind": "repository",
              "url": "https://github.com/digitalepidemiologylab/llm_crowd_experts_annotation",
              "evidence": "The data (list of tweets identifiers and summarised anonymised datasets) and R and Python code used can be found in the online repository at https://github.com/digitalepidemiologylab/llm_crowd_experts_annotation."
            },
            {
              "what": "Espinosa and Salathe applied the GPT-4 few-shot method this paper validated to 2197090 Portuguese tweets, classifying 1703009 by vaccination stance across Brazil 2013-2019 and comparing the resulting stance index against MMR coverage and measles cases",
              "kind": "successor-work",
              "url": "https://www.medrxiv.org/content/10.1101/2025.02.24.25322766v1.full-text",
              "evidence": "Following results from our previous research on the capacity of GPT-4 to assess vaccination-related tweets and its stable performance using different prompts, we chose the model to extract the stance towards vaccination from the tweets' text. 15"
            }
          ],
          "data_description": "1000 English-language tweets on vaccination, 2019-2022.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Persuasion & cognitive effects": [
              "Vaccine hesitancy",
              "Stance detection",
              "Public health surveillance"
            ]
          },
          "theme_qualifiers_canonical": {
            "Persuasion & cognitive effects": [
              "Vaccine hesitancy"
            ]
          },
          "tentative": false
        },
        {
          "title": "Clusters of science- and health-related Twitter users become more isolated during the COVID-19 pandemic",
          "wid": "clusters-of-science-and-health-related-twitter-users-be",
          "type": "publication",
          "year": 2021,
          "venue": "Nature Scientific Reports 11, 19655",
          "link": "https://doi.org/10.1038/s41598-021-99301-0",
          "authors": [
            "Francesco Durazzi (University of Bologna)",
            "Marcel Salathé (Digital Epidemiology Lab, EPFL)",
            "Martin Müller (Digital Epidemiology Lab, EPFL)",
            "Daniel Remondini (University of Bologna)"
          ],
          "epfl_authors": [
            "Marcel Salathé (Digital Epidemiology Lab, EPFL)",
            "Martin Müller (Digital Epidemiology Lab, EPFL)"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "COVID-19 infodemic",
            "Network analysis",
            "Scientific authority",
            "Social media communities",
            "Attention dynamics",
            "Polarisation"
          ],
          "stage": "Monitoring",
          "relevance": 3,
          "about": "Study mapping how attention to scientific and health voices on Twitter changed over the first months of the COVID-19 pandemic. The authors built the retweet network, ran community detection, and labeled accounts into occupational categories. Attention was extremely concentrated, with the top 0.1 percent of users receiving 77 percent of all retweets. Early on, the international science-health group drew disproportionate attention and reached across communities; from March onward attention shifted to national elites and political communities, the political group sustained far higher attention per user while amplifying itself internally, and the science-health group lost external attention and grew more isolated.",
          "why": "It quantifies how scientific and health voices lost ground to political communities during the pandemic infodemic.",
          "data": "Text, 353993900 English-language tweets from 26262332 users (Jan 13 to Jun 7, 2020)",
          "themes": [
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Echo chambers",
            "Exposure intensity",
            "Political polarisation"
          ],
          "key_terms": [
            "COVID-19",
            "Retweet networks",
            "Science communication",
            "Twitter",
            "Community detection"
          ],
          "models": [
            "BERT",
            "roBERTa"
          ],
          "method_qualifiers": [
            "Sentiment analysis",
            "Community detection"
          ],
          "events_cases": [
            "COVID-19 pandemic"
          ],
          "built_at_epfl": [
            {
              "name": "COVID-19 Twitter data, keyword stream 2020-01-13 to 2020-06-06",
              "kind": "dataset",
              "url": "https://doi.org/10.5281/zenodo.4267033",
              "evidence": "Paper, Data availability: \"The full Twitter dataset used in this work is available on Zenodo 45: https://doi.org/10.5281/zenodo.4267033.\" The Zenodo record's own title is verbatim \"COVID-19 Twitter data, keyword stream 2020-01-13 to 2020-06-06\", deposited by Mueller, Martin (EPFL); Durazzi, Francesco (University of Bologna); Remondini, Daniel (University of Bologna); Salathe, Marcel (EPFL), and its description gives the same counts as the paper: \"353993900 tweets (thereof 267026740 retweets) posted by 26262332 users\", collected via the Twitter filter streaming endpoint through the Crowdbreaks platform."
            }
          ],
          "follow_up": [
            {
              "what": "twitter-network-covid19, the public analysis code and aggregated data released with the paper (figure scripts, notebooks, community/network analysis)",
              "kind": "repository",
              "url": "https://github.com/FraDurazzi/twitter-network-covid19",
              "evidence": "Paper, Data availability: \"The datasets and the code generated and/or analysed during the current stury are available through the public GitHub repository https://github.com/FraDurazzi/twitter-network-covid19.\" The repository README states: \"This repository contains data and code in order to reproduce the results from the study 'International expert communities on Twitter become more isolated during the COVID-19 pandemic'.\" and \"Twitter data related to this study can be found here [https://zenodo.org/record/4267033]\"."
            }
          ],
          "data_description": "354M tweets, 26M users, Jan-June 2020.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "region_country": [
            "Global"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Spread, amplification & networks": [
              "Echo chambers",
              "Political polarisation",
              "Exposure intensity"
            ]
          },
          "theme_qualifiers_canonical": {
            "Spread, amplification & networks": [
              "Echo chambers",
              "Exposure intensity",
              "Political polarisation"
            ]
          },
          "tentative": false
        },
        {
          "title": "EPFL AI Citizen Assembly on AI (2025)",
          "wid": "epfl-ai-citizen-assembly-on-ai-2025",
          "type": "event",
          "year": 2025,
          "venue": "EPFL AI Center + Pôle de recherche en innovations démocratiques (UNIGE) + Demoscan",
          "link": "https://ai.epfl.ch/innovation/citizens-assembly-on-ai/",
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "AI governance",
            "Switzerland",
            "Deliberative democracy",
            "Deepfakes",
            "Disinformation",
            "Federal AI office"
          ],
          "stage": "NA",
          "relevance": 3,
          "about": "Deliberative-democracy initiative in which a representative survey of French-speaking Switzerland on attitudes toward AI fed into a 40-citizen assembly that deliberated AI governance and produced 20 concrete proposals across five areas, including content traceability. In the survey, deepfakes and disinformation were a top public concern at 77 percent. Among the proposals are a federal AI office and a labeling scheme to identify and promote human-made content, both responses to the difficulty of telling real information from AI-generated content.",
          "why": "Deepfakes and disinformation were a top public concern, and the assembly proposed content-traceability labeling as a direct response.",
          "data": "not applicable (deliberative initiative)",
          "lab": "EPFL AI Center",
          "themes": [
            "Media literacy & public resilience",
            "Platform governance & regulation",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Copyright compliance",
            "Critical thinking support",
            "Digital literacy",
            "Media provenance",
            "Public awareness campaigns",
            "Technical standards"
          ],
          "key_terms": [
            "Citizens' assembly",
            "Deliberative democracy",
            "AI governance",
            "Deepfakes",
            "Disinformation"
          ],
          "models": [],
          "method_qualifiers": [
            "Thematic analysis"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "follow_up": [
            {
              "what": "Final report of the Citizens' Assembly on artificial intelligence, published February 2026, carrying the 20 citizen proposals and the survey of 734 respondents",
              "kind": "successor-work",
              "url": "https://ai.epfl.ch/wp-content/uploads/Citizens_Assembly_AI_2025_ENG_.pdf",
              "evidence": "Title page: \"Citizens' Assembly on artificial intelligence / FRENCH-SPEAKING SWITZERLAND, 2025 / Final report Survey results and citizens' proposals / 40 citizens - 734 respondents - 20 proposals / February 2026\". Executive summary: \"This report presents the results of the Citizens' Assembly on artificial intelligence, organized by the EPFL AI Center in collaboration with the Universite de Geneve in November 2025.\" Marcel Salathe is named in the proceedings (\"an expert presentation given by Professor Marcel Salathe (EPFL AI Center)\") and in the Acknowledgements (\"Prof. Marcel Salathe, EPFL AI Center\")."
            },
            {
              "what": "Twenty citizen proposals for Swiss AI governance, structured around five issues including a federal AI office and the labelling of human-made content",
              "kind": "successor-work",
              "url": "https://ai.epfl.ch/innovation/citizens-assembly-on-ai",
              "evidence": "Report executive summary: \"After four days of deliberation, the 40 participants formulated 20 concrete proposals structured around five major issues: Issue 1 - The role of the state: creation of a federal AI office, long-term funding for research. Issue 2 - Access and education... Issue 3 - The world of work... Issue 4 - Traceability: labeling human creation, strengthening copyright. Issue 5 - Responsible practices: ethical legislation, fighting cybercrime.\" The EPFL AI Center page (opened) hosts the report in French, German and English and describes the assembly as \"organized in november 2025 by the EPFL AI Center in collaboration with the Research Hub on Democratic Innovations at the University of Geneva and Demoscan\"."
            }
          ],
          "data_description": "734 survey responses and transcripts from 40 citizens over four days.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "Switzerland"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Media literacy & public resilience": [
              "Digital literacy",
              "Public awareness campaigns",
              "Critical thinking support"
            ],
            "Platform governance & regulation": [
              "Copyright compliance",
              "Technical standards"
            ],
            "Verification & content authenticity": [
              "Media provenance"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media literacy & public resilience": [
              "Critical thinking support",
              "Digital literacy",
              "Public awareness campaigns"
            ],
            "Platform governance & regulation": [
              "Copyright compliance",
              "Regulatory oversight & audit",
              "Technical standards"
            ],
            "Verification & content authenticity": [
              "AI-generated content disclosure",
              "Media provenance"
            ]
          },
          "tentative": true
        },
        {
          "title": "IA, comment ne pas perdre le nord ?",
          "wid": "ia-comment-ne-pas-perdre-le-nord",
          "type": "other",
          "year": 2025,
          "venue": "Quanto (EPFL Press)",
          "link": "https://www.epflpress.org/produit/1612/9782889157280/ia-comment-ne-pas-perdre-le-nord",
          "authors": [
            "Marcel Salathé"
          ],
          "epfl_authors": [
            "Marcel Salathé"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "AI literacy",
            "Deepfakes",
            "Public understanding of AI",
            "Critical thinking"
          ],
          "stage": "NA",
          "relevance": 2,
          "about": "General-audience essay on how artificial intelligence is reshaping society. It traces AI from its earliest roots to today's large-scale models and explains the underlying ideas in plain terms: how neural networks work, how images and text get generated, the infrastructure behind these systems, and the effects on employment. It gives attention to deepfakes and synthetic media and to how people can keep reading AI-driven information critically.",
          "why": "It builds public AI literacy that helps readers recognise synthetic media and judge AI-generated content.",
          "data": "not applicable (book)",
          "lab": "Marcel Salathé",
          "themes": [],
          "subtopics": [],
          "key_terms": [
            "Deepfakes",
            "Synthetic media",
            "Neural networks",
            "Superintelligence",
            "AI and society"
          ],
          "models": [],
          "method_qualifiers": [],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "Book, 256 pages, publisher description.",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {},
          "theme_qualifiers_canonical": {},
          "tentative": true
        },
        {
          "title": "Decentralized Social Media and Artificial Intelligence in Digital Public Health Monitoring",
          "wid": "decentralized-social-media-and-artificial-intelligence-",
          "type": "publication",
          "year": 2025,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2512.04232",
          "authors": [
            "Marcel Salathé",
            "Sharada P. Mohanty"
          ],
          "epfl_authors": [
            "Marcel Salathé",
            "Sharada P. Mohanty"
          ],
          "about": "Viewpoint on how digital public health monitoring navigates two countervailing trends: platform policy changes restricting social media data access, and large language models expanding the capacity to analyze text. It discusses Mastodon and Bluesky as alternative data sources, noting a vaccine misinformation outbreak might rage on one platform but be absent on another. A pilot analysis of over 90 million Mastodon posts with GPT-4o mini shows standard digital epidemiology signals remain observable; the authors call these results preliminary and say further validation against ground-truth epidemiological data is needed. They argue for embracing new platforms, focusing on common diseases and broad signals, and advocating policies that preserve researchers' access to public data in privacy-respecting ways.",
          "themes": [
            "Platform governance & regulation",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "API pricing changes",
            "Cross-platform migration",
            "EU DSA data access",
            "Echo chambers",
            "Technical standards"
          ],
          "mdh_focus": [
            "M"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "Monitoring",
          "key_terms": [
            "Digital public health surveillance",
            "Vaccine misinformation",
            "Decentralized social media",
            "Mastodon",
            "API access restrictions"
          ],
          "models": [
            "GPT-4o mini"
          ],
          "method_qualifiers": [
            "Sentiment analysis",
            "Topic modeling",
            "Zero-shot prompting",
            "Structured output extraction"
          ],
          "events_cases": [
            "COVID-19 pandemic",
            "Twitter API shutdown",
            "Twitter free API access withdrawal (2023)"
          ],
          "built_at_epfl": [],
          "data_description": "90 million Mastodon posts, September 2023 to October 2024.",
          "platform": [
            "Bluesky",
            "Mastodon",
            "Twitter/X"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Platform governance & regulation": [
              "EU DSA data access",
              "API pricing changes",
              "Technical standards"
            ],
            "Spread, amplification & networks": [
              "Echo chambers",
              "Cross-platform migration"
            ]
          },
          "theme_qualifiers_canonical": {
            "Platform governance & regulation": [
              "Technical standards"
            ],
            "Spread, amplification & networks": [
              "Cross-platform migration",
              "Echo chambers"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "carmela",
      "name": "Carmela Troncoso",
      "url": "https://people.epfl.ch/carmela.troncoso",
      "unit": "SPRING Lab",
      "faculty": "IC (also MPI-SP, Bochum)",
      "mdh_focus": [
        "D",
        "M",
        "H"
      ],
      "dataTypes": [
        "Text",
        "Image"
      ],
      "techTypes": [
        "Privacy Engineering",
        "Propaganda Detection",
        "Concept Filtering",
        "Cryptography",
        "Decentralized ML Security"
      ],
      "stage": "Prevention + Monitoring",
      "publications": [
        {
          "title": "Characterizing and Detecting Propaganda-Spreading Accounts on Telegram",
          "wid": "characterizing-and-detecting-propaganda-spreading-accou",
          "type": "publication",
          "year": 2025,
          "venue": "34th USENIX Security Symposium (USENIX Security 2025)",
          "link": "https://www.usenix.org/conference/usenixsecurity25/presentation/kireev",
          "authors": [
            "Klim Kireev (EPFL / MPI-SP)",
            "Yevhen Mykhno (independent)",
            "Carmela Troncoso (EPFL / MPI-SP)",
            "Rebekah Overdorf (UNIL / Ruhr University Bochum)"
          ],
          "epfl_authors": [
            "Klim Kireev (EPFL / MPI-SP)",
            "Carmela Troncoso (EPFL / MPI-SP)",
            "Rebekah Overdorf (UNIL / Ruhr University Bochum)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Propaganda",
            "Telegram",
            "Coordinated accounts",
            "Russia-Ukraine",
            "Detection",
            "Content moderation",
            "Telegram propaganda detection",
            "Russian information operations"
          ],
          "stage": "Monitoring + Mitigation",
          "relevance": 5,
          "about": "Study of how propaganda operates on Telegram, a platform where messages appear chronologically and moderation is left to channel owners. It builds a new labeled dataset of messages from political and news channels in Russian, Belarusian and Ukrainian, combining a long historical export with real-time capture so deleted messages are preserved. It surfaces two independent coordinated networks, one pro-Russian and one pro-Ukrainian, whose propaganda messages drew about as many replies as messages from real users. Using only the information a moderator can see, it builds a classifier that detects propaganda from a single observed message, reaching 97.4 percent accuracy and holding up on unseen topics. It received a Distinguished Paper Award at USENIX Security 2025.",
          "why": "It maps two coordinated disinformation networks and delivers a propaganda detector.",
          "data": "Text, 17.3M messages from 13 political/news channels (Russian, Belarusian, Ukrainian)",
          "themes": [
            "Content moderation & enforcement",
            "Influence operations & coordinated manipulation",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Account coordination graphs",
            "Channel-level moderation",
            "Coordinated propaganda networks",
            "Moderator-assist detection",
            "Tactical reframing"
          ],
          "key_terms": [
            "Russo-Ukrainian war",
            "Propaganda accounts",
            "Instant-messaging platforms",
            "Propaganda detection",
            "Coordinated account networks"
          ],
          "models": [
            "GPT-4",
            "SBERT",
            "XGBoost"
          ],
          "method_qualifiers": [
            "Adversarial evasion",
            "Supervised classification"
          ],
          "events_cases": [
            "Russo-Ukrainian war",
            "Wagner Group rebellion"
          ],
          "built_at_epfl": [
            {
              "name": "Telegram propaganda dataset",
              "kind": "dataset",
              "url": "https://zenodo.org/records/14736756",
              "evidence": "Paper, Contributions: \"We compile the first labeled Telegram propaganda dataset of group messages and channel comments. This dataset comprises 17.3M labeled messages from 13 political and news-oriented channels. ... The dataset is available on Zenodo 4.\" and, in the Data availability statement: \"Our dataset is published on Zenodo 4.\" The footnote prints \"https://zenodo.org/records/14736756\". The Zenodo record itself is titled \"Real-time and Historical Telegram Dataset Annotated by Propaganda\", authored by Kireev, Klim (Ecole Polytechnique Federale de Lausanne), with Mykhno, Yevgen as contributor and Troncoso, Carmela and Overdorf, Rebekah as supervisors, licensed CC BY 4.0."
            }
          ],
          "follow_up": [
            {
              "what": "A Telegram Dataset of Propaganda and its Moderation, ICWSM 2025 (Proceedings of the International AAAI Conference on Web and Social Media, vol. 19 no. 1, pp. 2510-2518) - a dedicated dataset paper by the same four authors that publishes and documents the data collected for this work",
              "kind": "successor-work",
              "url": "https://ojs.aaai.org/index.php/ICWSM/article/view/35952",
              "evidence": "ICWSM paper, Dataset section: \"we identified in our prior work (Kireev et al. 2024)\"; Channel Selection: \"Second, smaller channels that we identified manually as having apparently automated propaganda activity as described in our prior work (Kireev et al. 2024).\"; Discussion: \"As we explored in prior work (Kireev et al. 2024), this dataset can be used to detect and remove propaganda from Telegram. We implemented a classifier to detect future propaganda based on the needs and abilities of channel owners.\" Its reference list resolves that citation as \"Kireev, K.; Mykhno, Y.; Troncoso, C.; and Overdorf, R. 2024. Characterizing and Detecting Propaganda-Spreading Accounts on Telegram. arXiv preprint arXiv:2406.08084.\", i.e. this work."
            }
          ],
          "data_description": "17.3M labeled Telegram messages from 13 channels, collected 2023.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Telegram"
          ],
          "region_country": [
            "Russia",
            "Ukraine"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Channel-level moderation",
              "Moderator-assist detection"
            ],
            "Influence operations & coordinated manipulation": [
              "Coordinated propaganda networks",
              "Inauthentic account behaviour",
              "Tactical reframing"
            ],
            "Spread, amplification & networks": [
              "Account coordination graphs"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Channel-level moderation",
              "Moderator-assist detection"
            ],
            "Influence operations & coordinated manipulation": [
              "Coordinated propaganda networks",
              "Narrative reframing",
              "Sockpuppet accounts"
            ],
            "Spread, amplification & networks": [
              "Account coordination graphs",
              "Cross-platform migration"
            ]
          },
          "tentative": false
        },
        {
          "title": "A Manually Annotated Image-Caption Dataset for Detecting Children in the Wild",
          "wid": "a-manually-annotated-image-caption-dataset-for-detectin",
          "type": "publication",
          "year": 2025,
          "venue": "arXiv 2506.10117 (submitted to NeurIPS 2025); MPI-SP and EPFL",
          "link": "https://arxiv.org/abs/2506.10117",
          "authors": [
            "Klim Kireev (MPI-SP / EPFL)",
            "Ana-Maria Creţu (EPFL)",
            "Raphael Meier (armasuisse S+T)",
            "Sarah Adel Bargal (Georgetown)",
            "Elissa Redmiles (Georgetown)",
            "Carmela Troncoso (MPI-SP / EPFL)"
          ],
          "epfl_authors": [
            "Klim Kireev (MPI-SP / EPFL)",
            "Ana-Maria Creţu (EPFL)",
            "Carmela Troncoso (MPI-SP / EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "H"
          ],
          "mdh_topics": [
            "CSAM",
            "AI-generated CSAM",
            "Minor detection",
            "T2I models",
            "Dataset filtering",
            "Benchmark"
          ],
          "stage": "Prevention + Monitoring",
          "relevance": 4,
          "about": "Releases ICCWD, the first image-caption dataset built to benchmark methods that detect depictions of minors. It contains manually labeled image-caption pairs and is richer than earlier child-image datasets, including fictional depictions such as sculptures, cartoons and drawings as well as partially visible bodies, with very high inter-annotator agreement. The motivation is the rise of AI-generated child sexual abuse material produced with text-to-image tools and proposals to filter children out of training sets. Testing off-the-shelf detectors showed the task is hard, with high detection coming at the cost of many false positives.",
          "why": "It provides the first benchmark for detecting minors, motivated by AI-generated CSAM and the need to filter children from training data.",
          "data": "Multimodal image+caption; ICCWD dataset of 10000 manually labelled pairs from Google CC3M (1675 child images, 8262 not-child, 63 disagreement)",
          "themes": [
            "Content moderation & enforcement",
            "Online sexual abuse & image-based abuse"
          ],
          "subtopics": [
            "AI-generated CSAM",
            "Automated pre-filtering",
            "Child image detection",
            "Nudify services"
          ],
          "key_terms": [
            "AI-generated CSAM",
            "Minor detection",
            "CSAM",
            "AIG-CSAM",
            "Image-caption dataset"
          ],
          "models": [
            "Amazon Rekognition Image",
            "DeepSeek-V3",
            "YOLO-11"
          ],
          "method_qualifiers": [
            "Manual annotation",
            "Zero-shot prompting",
            "Face-based age estimation"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Image-Caption Children in the Wild Dataset (ICCWD)",
              "kind": "dataset"
            },
            {
              "name": "ICCWD",
              "kind": "dataset",
              "url": "https://huggingface.co/datasets/amcretu/iccwd",
              "evidence": "Paper, Contributions paragraph: \"We make our dataset publicly available on HuggingFace 2 .\" with footnote \"2 https://huggingface.co/datasets/amcretu/iccwd\". The dataset card on that page carries the BibTeX for \"A Manually Annotated Image-Caption Dataset for Detecting Children in the Wild\" (Kireev, Cretu, Meier, Bargal, Redmiles, Troncoso, arXiv:2506.10117)."
            }
          ],
          "follow_up": [
            {
              "what": "spring-epfl/iccwd, the benchmark code released with the paper (image download from the released URLs, plus the DeepSeek-V3, Amazon Rekognition and combined minor-detection evaluations)",
              "kind": "repository",
              "url": "https://github.com/spring-epfl/iccwd",
              "evidence": "Repo README: \"# Code for the Image-Caption in the Wild Dataset (ICCWD)\" ... \"We provide the code to download images of ICCWD and evaluate three different minor detection methods on it.\" ... \"The dataset is available on HuggingFace as this [link](https://huggingface.co/datasets/amcretu/iccwd).\" The paper points at it: \"We make our code available at GitHub 3 .\" with footnote \"3 https://github.com/spring-epfl/iccwd/\"."
            }
          ],
          "data_description": "10000 image-caption pairs from CC3M, 1675 labeled Child.",
          "follow_up_checked": "2026-09-02",
          "targeted_group": [
            "Children",
            "Minors"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Automated pre-filtering",
              "Moderator-assist detection"
            ],
            "Online sexual abuse & image-based abuse": [
              "AI-generated CSAM",
              "Child image detection",
              "Nudify services"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Automated pre-filtering",
              "Moderator-assist detection"
            ],
            "Online sexual abuse & image-based abuse": [
              "AI-generated CSAM",
              "Child image detection",
              "Nudify services"
            ]
          },
          "tentative": false
        },
        {
          "title": "A Telegram Dataset of Propaganda and its Moderation",
          "wid": "a-telegram-dataset-of-propaganda-and-its-moderation",
          "type": "publication",
          "year": 2025,
          "venue": "ICWSM 2025 (Nineteenth International AAAI Conference on Web and Social Media)",
          "link": "https://ojs.aaai.org/index.php/ICWSM/article/view/35952",
          "authors": [
            "Klim Kireev (EPFL / MPI-SP)",
            "Yevhen Mykhno (independent)",
            "Carmela Troncoso (EPFL / MPI-SP)",
            "Rebekah Overdorf (UNIL / Ruhr University Bochum)"
          ],
          "epfl_authors": [
            "Klim Kireev (EPFL / MPI-SP)",
            "Carmela Troncoso (EPFL / MPI-SP)",
            "Rebekah Overdorf (UNIL / Ruhr University Bochum)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Telegram",
            "Propaganda dataset",
            "Russia-Ukraine",
            "Deletion labels",
            "Real-time collection",
            "Companion to USENIX Sec 2025"
          ],
          "stage": "Monitoring + Mitigation",
          "relevance": 5,
          "about": "Dataset paper releasing a large corpus of Telegram messages for studying propaganda and platform moderation. It collects messages from channels in Russian, Belarusian and Ukrainian using both the export API for years of history and a real-time client, which together capture messages later deleted by moderators or users. Messages carry labels for deletion and for membership in pro-Russian or pro-Ukrainian propaganda networks, with channels spanning right-wing to neutral stances. It is released openly as a companion to the lab's propaganda-detection paper, on a platform the authors note is understudied relative to its 950M monthly users.",
          "why": "It is a labeled disinformation and propaganda corpus built to support misinformation and moderation research.",
          "data": "Text; 17.3M Telegram messages from 13 channels (Russian, Belarusian, Ukrainian); dual collection (export API for 36-month history + real-time client Aug 16 - Oct 16 2023); labels for deletion + pro-Russian / pro-Ukrainian propaganda network membership; on Zenodo",
          "themes": [
            "Content moderation & enforcement",
            "Influence operations & coordinated manipulation"
          ],
          "subtopics": [
            "Coordinated reply attacks",
            "Duplicated message templates",
            "Inauthentic account behaviour",
            "Moderation selectivity"
          ],
          "key_terms": [
            "Russo-Ukrainian war",
            "Propaganda networks",
            "Telegram",
            "Content moderation",
            "Coordinated inauthentic replies"
          ],
          "models": [
            "SBERT"
          ],
          "method_qualifiers": [
            "Topic modeling",
            "Manual annotation",
            "Snowball sampling"
          ],
          "events_cases": [
            "Russo-Ukrainian war"
          ],
          "built_at_epfl": [
            {
              "name": "Telegram propaganda and moderation dataset",
              "kind": "dataset",
              "url": "https://zenodo.org/records/14661891",
              "evidence": "Paper: \"The dataset follows FAIR principles (Wilkinson et al. 2016), since it is published on a popular platform 2 (Findable), openly accessible, stored in user-friendly CSV format (Interpretable), under a Creative Commons license (Reusable).\", with footnote \"2 https://zenodo.org/records/14661891\". The Zenodo record is titled \"Real-time and Historical Telegram Dataset Annotated by Propaganda\", author \"Klim Kireev (Ecole Polytechnique Federale de Lausanne)\", contributors \"Yevgen Mykhno (Annotator)\" and \"Carmela Troncoso and Rebekah Overdorf (Supervisors)\", licensed \"Creative Commons Attribution 4.0 International\", published 16 January 2025."
            }
          ],
          "data_description": "17.3M Telegram messages from 13 channels, 2021-2023.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Telegram"
          ],
          "region_country": [
            "Russia",
            "Ukraine"
          ],
          "targeted_group": [
            "Ethnic minorities"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Moderation selectivity"
            ],
            "Influence operations & coordinated manipulation": [
              "Duplicated message templates",
              "Coordinated reply attacks",
              "Inauthentic account behaviour"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Moderation selectivity"
            ],
            "Influence operations & coordinated manipulation": [
              "Coordinated reply attacks",
              "Duplicated message templates",
              "Sockpuppet accounts"
            ]
          },
          "tentative": false
        },
        {
          "title": "Evaluating Concept Filtering Defenses against Child Sexual Abuse Material Generation by Text-to-Image Models",
          "wid": "evaluating-concept-filtering-defenses-against-child-sex",
          "type": "publication",
          "year": 2026,
          "venue": "IEEE Symposium on Security and Privacy 2026 (extended arXiv 2512.05707)",
          "link": "https://arxiv.org/abs/2512.05707",
          "authors": [
            "Ana-Maria Creţu (EPFL)",
            "Klim Kireev (EPFL / MPI-SP)",
            "Amro Abdalla (Georgetown)",
            "Wisdom Obinna (Georgetown)",
            "Raphael Meier (armasuisse S+T)",
            "Sarah Adel Bargal (Georgetown)",
            "Elissa M. Redmiles (Georgetown)",
            "Carmela Troncoso (EPFL / MPI-SP)"
          ],
          "epfl_authors": [
            "Klim Kireev (EPFL / MPI-SP)",
            "Carmela Troncoso (EPFL / MPI-SP)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "H"
          ],
          "mdh_topics": [
            "CSAM",
            "AI-generated CSAM",
            "Concept filtering",
            "T2I model security",
            "Adversarial prompting",
            "Fine-tuning attack",
            "Cryptographic security game"
          ],
          "stage": "Prevention",
          "relevance": 5,
          "about": "Study testing whether removing child images from a text-to-image model's training data actually stops the model from producing AI-generated child sexual abuse material. It formalises the problem as a security game and uses 'child wearing glasses' as an ethical and legal proxy, testing many automated child-detection methods and training models from scratch on filtered data before probing them with prompting and fine-tuning. The best detector still leaves a small fraction of child images, enough that an adversary needs only a handful of prompts to generate the target image, and fine-tuning on a small set of child images cancels the filtering. The authors conclude that current child-filtering defenses give limited protection to closed-weight models and none to open-weight models.",
          "why": "It evaluates a safety filter meant to prevent AI-generated child sexual abuse imagery and shows the defense fails.",
          "data": "Multimodal image+caption; 20+ child detection methods tested on CC3M + LAION-Face image-caption datasets; Stable Diffusion 1.x trained from scratch on filtered datasets; user-study evaluation of generated images",
          "themes": [
            "AI safety",
            "Online sexual abuse & image-based abuse"
          ],
          "subtopics": [
            "AI-generated CSAM",
            "Adaptive attacks",
            "Child image detection",
            "Misuse risk assessment",
            "Training-corpus design"
          ],
          "key_terms": [
            "Text-to-image models",
            "Training data filtering",
            "Stable Diffusion",
            "AI-generated CSAM",
            "AIG-CSAM"
          ],
          "models": [
            "Amazon Rekognition",
            "DeepSeek-V3",
            "FairFace",
            "LLaVA-7B",
            "MiVOLO",
            "Stable Diffusion"
          ],
          "method_qualifiers": [
            "Game-based security definition",
            "LoRA fine-tuning",
            "Black-box auditing",
            "Adversarial evasion"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "AIG-CSAM security game",
              "kind": "framework"
            },
            {
              "name": "t2i-child-filtering",
              "kind": "tool"
            },
            {
              "name": "LAION-Face-2k",
              "kind": "dataset",
              "evidence": "Paper sec. 3: \"To benchmark methods on LAION-Face, we follow Kireev et al.'s methodology[17] to label 2000 images of LAION-Face with child/no child labels using two authors as the annotators. [...] We refer to this dataset as LAION-Face-2k.\" It is deliberately unpublished. Appendix A (Ethics considerations), verbatim: \"We do not publicly release the LAION-Face subset we have manually annotated with child labels to prevent bad actors from using it for validation.\" No url, by the authors' own design."
            }
          ],
          "follow_up": [
            {
              "what": "Source-code repository for this paper: spring-epfl/t2i-child-filtering on GitHub, created 2026-03-13. Currently a placeholder holding only a 160-byte README; no code had been pushed as of 2026-09-02.",
              "kind": "repository",
              "url": "https://github.com/spring-epfl/t2i-child-filtering",
              "evidence": "Paper, footnote 1 on page 1, verbatim: \"Source code: https://github.com/spring-epfl/t2i-child-filtering.\" The repository's own GitHub description reads: \"Source code for the 'Evaluating Concept Filtering Defenses against Child Sexual Abuse Material Generation by Text-to-Image Models' paper.\""
            }
          ],
          "data_description": "CC3M (2.3M) and LAION-Face (34.3M) image-caption pairs; 1135 Prolific raters.",
          "follow_up_checked": "2026-09-02",
          "targeted_group": [
            "Children"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Training-corpus design",
              "Adaptive attacks",
              "Misuse risk assessment"
            ],
            "Online sexual abuse & image-based abuse": [
              "AI-generated CSAM",
              "Child image detection"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Adaptive attacks",
              "Misuse risk assessment",
              "Training-corpus design"
            ],
            "Online sexual abuse & image-based abuse": [
              "AI-generated CSAM",
              "Child image detection"
            ]
          },
          "tentative": false
        },
        {
          "title": "Bugs in our Pockets: The Risks of Client-Side Scanning",
          "wid": "bugs-in-our-pockets-the-risks-of-client-side-scanning",
          "type": "publication",
          "year": 2024,
          "venue": "Journal of Cybersecurity 10(1)",
          "link": "https://arxiv.org/abs/2110.07450",
          "authors": [
            "Hal Abelson",
            "Ross Anderson",
            "Steven M. Bellovin",
            "Josh Benaloh",
            "Matt Blaze",
            "Jon Callas",
            "Whitfield Diffie",
            "Susan Landau",
            "Peter G. Neumann",
            "Ronald L. Rivest",
            "Jeffrey I. Schiller",
            "Bruce Schneier",
            "Vanessa Teague",
            "Carmela Troncoso"
          ],
          "epfl_authors": [
            "Carmela Troncoso"
          ],
          "about": "Report analysing client-side scanning (CSS), a proposal to scan material on users' devices before it is encrypted or after it is decrypted and alert agencies when targeted information is detected. It argues that CSS neither guarantees efficacious crime prevention nor prevents surveillance, detailing how perceptual hashing and machine learning scanners can be evaded, flooded with false alarms, and abused by governments, unauthorized parties and local adversaries. It also analyses Apple's August 2021 proposal, concluding that Apple has still not produced a secure and trustworthy design, and the authors find no design space for solutions that provide substantial benefits to law enforcement without unduly risking the privacy and security of law-abiding citizens.",
          "themes": [
            "Content moderation & enforcement",
            "Online sexual abuse & image-based abuse",
            "Platform governance & regulation"
          ],
          "subtopics": [
            "Automated pre-filtering",
            "CSAM hash lists",
            "Detection evasion",
            "State-ordered takedowns",
            "Technical standards"
          ],
          "mdh_focus": [
            "H"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "Monitoring + Mitigation",
          "key_terms": [
            "Client-side scanning",
            "CSAM detection",
            "Perceptual hashing",
            "End-to-end encryption",
            "CSAM"
          ],
          "models": [],
          "method_qualifiers": [
            "Threat modeling",
            "Algorithmic auditing"
          ],
          "events_cases": [
            "Apple's August 2021 CSAM scanning proposal"
          ],
          "built_at_epfl": [],
          "data_description": "No empirical data (policy and security analysis).",
          "targeted_group": [
            "Children",
            "LGBTQ+",
            "Political dissidents"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Automated pre-filtering",
              "State-ordered takedowns",
              "Detection evasion"
            ],
            "Online sexual abuse & image-based abuse": [
              "Child image detection",
              "CSAM hash lists"
            ],
            "Platform governance & regulation": [
              "Technical standards"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Automated pre-filtering",
              "Ban evasion",
              "State-ordered takedowns"
            ],
            "Online sexual abuse & image-based abuse": [
              "Child image detection",
              "Harms to depicted victims"
            ],
            "Platform governance & regulation": [
              "Online anonymity",
              "Regulatory oversight & audit",
              "Technical standards"
            ]
          },
          "tentative": false
        },
        {
          "title": "Neural Exec: Learning (and Learning from) Execution Triggers for Prompt Injection Attacks",
          "wid": "neural-exec-learning-and-learning-from-execution-trigge",
          "type": "publication",
          "year": 2024,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2403.03792",
          "authors": [
            "Dario Pasquini",
            "Martin Strohmeier",
            "Carmela Troncoso"
          ],
          "epfl_authors": [
            "Dario Pasquini",
            "Carmela Troncoso"
          ],
          "about": "Paper introducing Neural Exec, a family of prompt injection attacks that treats creating execution triggers as a differentiable search problem, instead of relying on handcrafted strings such as Ignore previous instructions. Across four open-source LLMs, they are at least twice as effective as the best baseline trigger, with an average execution accuracy of 91 percent. They can be designed to persist through Retrieval-Augmented Generation pipelines, with an average persistence of 80 percent at the usual default chunk size of 500 characters. Because they deviate markedly in form from any known attack, such triggers can sidestep existing blacklist-based detection and sanitation approaches.",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Adaptive attacks",
            "Adversarial prompting",
            "Retrieval pipeline poisoning"
          ],
          "mdh_focus": [
            "D"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "NA",
          "key_terms": [
            "Prompt injection attacks",
            "Execution triggers",
            "Retrieval-Augmented Generation",
            "LLM security",
            "Prompt injection"
          ],
          "models": [
            "ChatGPT-4",
            "ChatGPT4",
            "Llama-3-8B-Instruct",
            "Mistral-7B-Instruct-v0.2",
            "Mixtral-8x7B",
            "OpenChat3.5"
          ],
          "method_qualifiers": [
            "Prompt injection",
            "Adversarial evasion",
            "Greedy Coordinate Gradient"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "LLM_NeuralExec",
              "kind": "tool"
            },
            {
              "name": "Neural Exec",
              "kind": "framework"
            }
          ],
          "data_description": "Generated prompts from Alpaca, SQuAD 2.0 and Code Alpaca; 100-prompt test set.",
          "platform": [
            "HuggingChat"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Adaptive attacks",
              "Adversarial prompting",
              "Retrieval pipeline poisoning"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Adaptive attacks"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "vandergheynst",
      "name": "Pierre Vandergheynst",
      "url": "https://people.epfl.ch/pierre.vandergheynst",
      "unit": "LTS2",
      "faculty": "STI",
      "mdh_focus": [
        "D",
        "M"
      ],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "Graph Signal Processing",
        "Network Analysis",
        "Deep Learning",
        "Bot Detection"
      ],
      "stage": "Monitoring",
      "publications": [
        {
          "title": "Social Network and Architectures of Disinformation",
          "wid": "social-network-and-architectures-of-disinformation",
          "type": "thesis",
          "year": 2020,
          "venue": "EPFL Master Thesis (Electrical Engineering), project performed at Radio Télévision Suisse",
          "link": "https://infoscience.epfl.ch/record/280275",
          "authors": [
            "Amin Mekacher"
          ],
          "epfl_authors": [
            "Amin Mekacher"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_topics": [
            "Conspiracy theories",
            "Reddit",
            "Voat",
            "QAnon",
            "Coronavirus disinformation",
            "Bot networks"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "Master thesis, carried out with Radio Television Suisse, mapping the internal structure of conspiratorial and alt-right online communities to find markers that distinguish them from ordinary forums. It gathers data from Reddit, Twitter and the uncensored site Voat, and runs three analyses: how the most rewarded r/conspiracy users migrate across related subreddits (with r/science as a control), which sources were cited in Reddit's coronavirus discussion during the first COVID-19 wave, and the network structure of QAnon communities on Voat. It finds that top r/conspiracy users form a tightly connected core that recirculates across related subreddits, that pandemic narratives such as 5G, Wuhan and adrenochrome spike together, and that QAnon communities on Voat are almost entirely inward-facing.",
          "why": "It maps disinformation and conspiracy communities such as QAnon and COVID-19 narratives across Reddit and Voat.",
          "data": "Text, Graph/Network (Reddit, Voat, some Twitter linking)",
          "themes": [
            "Influence operations & coordinated manipulation",
            "Radicalisation & violent extremism",
            "Spread, amplification & networks",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Bot amplification",
            "Coordinated propaganda networks",
            "Cross-platform migration",
            "Far-right communities",
            "Radicalisation pathways",
            "Sockpuppet accounts"
          ],
          "key_terms": [
            "QAnon",
            "Conspiracy theories",
            "Alt-right",
            "Voat",
            "Bots"
          ],
          "models": [],
          "method_qualifiers": [
            "Collective influence ranking",
            "Web scraping",
            "Community detection",
            "Retrospective archive mining"
          ],
          "events_cases": [
            "COVID-19 pandemic",
            "Coronavirus outbreak",
            "QAnon movement"
          ],
          "built_at_epfl": [],
          "follow_up": [
            {
              "what": "reddit-conspiracy-analysis, the GitLab repository holding all Python code written for this thesis (created 2020-03-01, 19 commits)",
              "kind": "repository",
              "url": "https://gitlab.com/AminMekacher/reddit-conspiracy-analysis",
              "evidence": "The thesis itself states, in section 1 Project Introduction: 'The Python code developed throughout the project can be found at the following Gitlab repository.' The hyperlink behind that sentence resolves to https://gitlab.com/AminMekacher/reddit-conspiracy-analysis (extracted from the /URI annotations of the thesis PDF; it is the only GitLab link in the document). The GitLab project page shows 'reddit-conspiracy-analysis', 'Created on March 01, 2020', '19 Commits', '1 Branch'."
            },
            {
              "what": "reddit-graph-visualization, the GitHub Pages repository publishing the thesis's five interactive Gephi/sigma.js network graphs (RedditConspiracy, Anon2020, Voat2020, VoatComments, VoatReddit)",
              "kind": "repository",
              "url": "https://github.com/AminMekacher/reddit-graph-visualization",
              "evidence": "The thesis links five interactive graphs, e.g. 'If we use the detected communities as colors, we get the following network, which can also be found in an interactive version at this address', 'The resulting graph can be found in an interactive version at this web address', and 'An interactive version of the generated network can be found at this address'. The five hyperlink targets extracted from the PDF are https://aminmekacher.github.io/reddit-graph-visualization/{RedditConspiracy,Anon2020,Voat2020,VoatComments,VoatReddit}/ , all HTTP 200. The GitHub repository AminMekacher/reddit-graph-visualization contains exactly those five directories: Anon2020, RedditConspiracy, Voat2020, VoatComments, VoatReddit. Its RedditConspiracy/data.json holds the thesis's own Reddit data (nodes labelled dataisbeautiful, cvnews, coronavirusflorida, with attributes subs5g and subsChloroquine). The thesis acknowledgments credit 'Volodymyr for his comprehensive guide to publish interactive graphs from Gephi on Github.'"
            },
            {
              "what": "IMI funded a continuation project, #sad2, launched February 2021, of the Initiative for Media Innovation project #SAD that this thesis was produced within; #sad2 is again led with Prof. Pierre Vandergheynst's LTS2 at EPFL, with UniNE AJM and RTS",
              "kind": "project",
              "url": "https://www.media-initiative.ch/project/social-network-architectures-of-disinformation-sad/",
              "evidence": "The #SAD project page lists under 'Project outputs': '\"Social network Architectures of Disinformation\" Master's thesis of Amin Mekacher (login required)', names 'Principal investigator Prof. Pierre Vandergheynst (EPFL)' and the media partner 'RTS - Radio Television Suisse', and states: 'A continuation of this project was launched in February 2021 with renewed support from IMI.' The linked #sad2 page at lextechinstitute.ch names 'Prof. Pierre Vandergheynst, LTS2 laboratory at EPFL', 'Benjamin Ricaud', 'Nicolas Aspert' and RTS via 'Felix Tybalt', and describes it as 'Based on the prototype developed in the first phase of research'."
            }
          ],
          "data_description": "Reddit, Voat and Twitter posts from 2020, including 185500 collected tweets.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Reddit",
            "Twitter/X",
            "Voat",
            "YouTube"
          ],
          "region_country": [
            "United States"
          ],
          "targeted_group": [
            "Chinese people",
            "Muslims",
            "Political groups",
            "Racial and ethnic minorities",
            "Religious or ethnic minorities",
            "women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Influence operations & coordinated manipulation": [
              "Coordinated propaganda networks",
              "Bot amplification",
              "Sockpuppet accounts"
            ],
            "Radicalisation & violent extremism": [
              "Far-right communities",
              "Radicalisation pathways"
            ],
            "Spread, amplification & networks": [
              "Cross-platform migration"
            ]
          },
          "theme_qualifiers_canonical": {
            "Influence operations & coordinated manipulation": [
              "Bot amplification",
              "Coordinated propaganda networks",
              "Sockpuppet accounts"
            ],
            "Radicalisation & violent extremism": [
              "Far-right communities",
              "Radicalisation pathways",
              "User migration"
            ],
            "Spread, amplification & networks": [
              "Account coordination graphs",
              "Cross-platform migration",
              "Echo chambers"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "bosselut",
      "name": "Antoine Bosselut",
      "url": "https://people.epfl.ch/antoine.bosselut",
      "unit": "NLP",
      "faculty": "IC",
      "mdh_focus": [
        "M",
        "D",
        "H"
      ],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "NLP",
        "LLMs",
        "Content Filtering",
        "Fact Verification",
        "Adversarial ML",
        "LLM Alignment"
      ],
      "stage": "Prevention + Monitoring",
      "publications": [
        {
          "title": "Apertus: Democratizing Open and Compliant LLMs for Global Language Environments",
          "wid": "apertus-democratizing-open-and-compliant-llms-for-globa",
          "type": "publication",
          "year": 2025,
          "venue": "arXiv 2509.14233",
          "link": "https://arxiv.org/abs/2509.14233",
          "authors": [
            "Alejandro Hernández-Cano",
            "Alexander Hägele",
            "Allen Hao Huang",
            "Angelika Romanou",
            "Antoni-Joan Solergibert",
            "Barna Pasztor",
            "Bettina Messmer",
            "Dhia Garbaya",
            "Eduard Frank Ďurech",
            "Ido Hakimi",
            "Juan García Giraldo",
            "Mete Ismayilzada",
            "Negar Foroutan",
            "Skander Moalla",
            "Tiancheng Chen",
            "Vinko Sabolčec",
            "Yixuan Xu",
            "Michael Aerni",
            "Badr AlKhamissi",
            "Inés Altemir Mariñas",
            "Mohammad Hossein Amani",
            "Matin Ansaripour",
            "Ilia Badanin",
            "Harold Benoit",
            "Emanuela Boros",
            "Nicholas Browning",
            "Fabian Bösch",
            "Maximilian Böther",
            "Niklas Canova",
            "Camille Challier",
            "Clement Charmillot",
            "Jonathan Coles",
            "Jan Deriu",
            "Arnout Devos",
            "Lukas Drescher",
            "Daniil Dzenhaliou",
            "Maud Ehrmann",
            "Dongyang Fan",
            "Simin Fan",
            "Silin Gao",
            "Miguel Gila",
            "María Grandury",
            "Diba Hashemi",
            "Alexander Hoyle",
            "Jiaming Jiang",
            "Mark Klein",
            "Andrei Kucharavy",
            "Anastasiia Kucherenko",
            "Frederike Lübeck",
            "Roman Machacek",
            "Theofilos Manitaras",
            "Andreas Marfurt",
            "Kyle Matoba",
            "Simon Matrenok",
            "Henrique Mendonça",
            "Fawzi Roberto Mohamed",
            "Syrielle Montariol",
            "Luca Mouchel",
            "Sven Najem-Meyer",
            "Jingwei Ni",
            "Gennaro Oliva",
            "Matteo Pagliardini",
            "Elia Palme",
            "Andrei Panferov",
            "Léo Paoletti",
            "Marco Passerini",
            "Ivan Pavlov",
            "Auguste Poiroux",
            "Kaustubh Ponkshe",
            "Nathan Ranchin",
            "Javi Rando",
            "Mathieu Sauser",
            "Jakhongir Saydaliev",
            "Muhammad Ali Sayfiddinov",
            "Marian Schneider",
            "Stefano Schuppli",
            "Marco Scialanga",
            "Andrei Semenov",
            "Kumar Shridhar",
            "Raghav Singhal",
            "Anna Sotnikova",
            "Alexander Sternfeld",
            "Ayush Kumar Tarun",
            "Paul Teiletche",
            "Jannis Vamvas",
            "Xiaozhe Yao",
            "Hao Zhao",
            "Alexander Ilic",
            "Ana Klimovic",
            "Andreas Krause",
            "Caglar Gulcehre",
            "David Rosenthal",
            "Elliott Ash",
            "Florian Tramèr",
            "Joost VandeVondele",
            "Livio Veraldi",
            "Martin Rajman",
            "Thomas Schulthess",
            "Torsten Hoefler",
            "Antoine Bosselut",
            "Martin Jaggi",
            "Imanol Schlag"
          ],
          "epfl_authors": [
            "Alejandro Hernández-Cano",
            "Alexander Hägele",
            "Allen Hao Huang",
            "Angelika Romanou",
            "Antoni-Joan Solergibert",
            "Bettina Messmer",
            "Dhia Garbaya",
            "Eduard Frank Ďurech",
            "Juan García Giraldo",
            "Mete Ismayilzada",
            "Negar Foroutan",
            "Skander Moalla",
            "Vinko Sabolčec",
            "Yixuan Xu",
            "Badr AlKhamissi",
            "Inés Altemir Mariñas",
            "Mohammad Hossein Amani",
            "Matin Ansaripour",
            "Ilia Badanin",
            "Harold Benoit",
            "Emanuela Boros",
            "Camille Challier",
            "Clement Charmillot",
            "Daniil Dzenhaliou",
            "Maud Ehrmann",
            "Dongyang Fan",
            "Simin Fan",
            "Silin Gao",
            "María Grandury",
            "Diba Hashemi",
            "Jiaming Jiang",
            "Kyle Matoba",
            "Simon Matrenok",
            "Syrielle Montariol",
            "Luca Mouchel",
            "Sven Najem-Meyer",
            "Matteo Pagliardini",
            "Léo Paoletti",
            "Ivan Pavlov",
            "Auguste Poiroux",
            "Kaustubh Ponkshe",
            "Nathan Ranchin",
            "Mathieu Sauser",
            "Jakhongir Saydaliev",
            "Marco Scialanga",
            "Andrei Semenov",
            "Raghav Singhal",
            "Anna Sotnikova",
            "Ayush Kumar Tarun",
            "Paul Teiletche",
            "Hao Zhao",
            "Caglar Gulcehre",
            "Martin Rajman",
            "Antoine Bosselut",
            "Martin Jaggi"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_topics": [
            "Open-source LLM",
            "Data compliance",
            "Multilingual",
            "Toxic-content filtering",
            "Swiss AI Initiative",
            "Digital sovereignty"
          ],
          "stage": "Prevention + Monitoring",
          "relevance": 3,
          "about": "Technical report for Apertus, a fully open suite of large language models trained on web-scale multilingual data spanning more than 1800 languages, with around 40 percent of pretraining data non-English. It trains only on openly available data, respects robots.txt exclusions, and filters out non-permissive, toxic and personally identifiable content, with everything released under a permissive license so the pipeline can be audited. Released at 8B and 70B scales and trained on 15 trillion tokens, it approaches state-of-the-art results among fully open models on multilingual benchmarks while training only on compliant data.",
          "why": "Its toxic-content filtering and auditable, opt-out-respecting pretraining make it a trustworthy base model for downstream MDH work.",
          "data": "Text, 15 trillion tokens across 1800+ languages",
          "lab": [
            "NLP",
            "MLO",
            "CLAIRE"
          ],
          "themes": [
            "AI safety",
            "Platform governance & regulation",
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Constitutional AI",
            "Implicit hate speech",
            "Misuse risk assessment",
            "Model-generated toxicity",
            "Safety alignment"
          ],
          "key_terms": [
            "Data compliance",
            "Swiss AI Charter",
            "Constitutional AI",
            "Constitutional AI alignment",
            "EU AI Act compliance"
          ],
          "models": [
            "Apertus-70B-Instruct"
          ],
          "method_qualifiers": [
            "Goldfish loss",
            "Constitutional AI",
            "Preference optimisation",
            "QRPO alignment"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Apertus-70B-Instruct",
              "kind": "model"
            },
            {
              "name": "Apertus-sft-mixture",
              "kind": "dataset"
            },
            {
              "name": "Swiss AI Charter",
              "kind": "framework",
              "url": "https://www.apertus-ai.org/pages/charter/",
              "evidence": "The Apertus Charter defines alignment principles for AI systems developed under the Swiss AI Initiative, rooted in Switzerland's constitutional values and democratic traditions. (Previously referred to as the Swiss AI Charter) Version 1.0 - August 2025"
            },
            {
              "name": "SwitzerlandQA",
              "kind": "benchmark",
              "evidence": "Paper Appendix K: \"we test the model on its understanding of Switzerland's environment by developing a novel benchmark SwitzerlandQA specifically tailored to Switzerland's context. [...] The benchmark represents 26 cantons, with each canton having at least 200 questions, yielding 9167 unique items per language across domains and levels of granularity.\" The artefact list names it as swiss-ai/switzerland_qa, but no public URL was found - see notes."
            },
            {
              "name": "apertus-pretrain-toxicity",
              "kind": "model",
              "url": "https://huggingface.co/swiss-ai/apertus-pretrain-toxicity",
              "evidence": "Model card: \"Language specific toxicity classifiers in English, French, German, Italian, Spanish, Portuguese, Polish, Chinese and Dutch, trained on PleIAs/ToxicCommons and SWSR-SexComments datasets.\" and \"The classifier checkpoints with the best accuracy on the held-out validation set are further employed to annotate the toxicity scores on FineWeb-2 and FineWeb.\" The paper cites this exact URL as footnote 13 in the toxicity-filtering section."
            },
            {
              "name": "RealToxicityPrompts-Llama-Subsampled",
              "kind": "benchmark",
              "url": "https://huggingface.co/datasets/swiss-ai/realtoxicityprompts",
              "evidence": "Paper: \"To integrate it in our benchmark harness, we sub-sample it to 10% of its size and switch the toxicity classifier model to Llama-Guard-3-8B (Fedorov et al., 2024) to allow fully-contained execution. We release this subsample, [48] as well as the LLaMA-Guard-3-8B implementation. [49] The resulting benchmark, RealToxicityPromptsLlama-Subsampled...\" with footnote 48 = https://huggingface.co/datasets/swiss-ai/realtoxicityprompts/tree/main/realtoxicityprompts_small. The dataset page confirms two subsets, realtoxicityprompts_full (99.4k rows) and realtoxicityprompts_small (10k rows)."
            }
          ],
          "follow_up": [
            {
              "what": "Swisscom deployed Apertus on its sovereign Swiss AI Platform for business customers on the day of release",
              "kind": "deployment",
              "url": "https://www.swisscom.ch/en/about/news/2025/09/02-apertus.html",
              "evidence": "Swisscom is proud to be among the first to deploy this pioneering large language model on our sovereign Swiss AI Platform. [...] As of today, Swisscom business customers will be able to access the Apertus model via Swisscom's sovereign Swiss AI platform."
            },
            {
              "what": "The Public AI Inference Utility became the official international deployer of Apertus, serving it worldwide",
              "kind": "deployment",
              "url": "https://publicai.co/stories/apertus",
              "evidence": "Public AI is proud to be the official international deployer for Apertus. To support Apertus, we've allocated over 115000 GPU-hours spread across 20 clusters in 5+ countries - just for the month of September."
            },
            {
              "what": "Apertus v1.1, a family of distilled models trained from the Apertus-8B-2509 teacher described in this report",
              "kind": "model",
              "url": "https://huggingface.co/swiss-ai/Apertus-v1.1-4B-Instruct",
              "evidence": "For more details refer to the original Apertus [technical report](https://arxiv.org/abs/2509.14233) and the new Apertus [distillation technical report](https://arxiv.org/abs/2605.29128)."
            },
            {
              "what": "Apertus v1.5, a multimodal continuation of the models released in this report",
              "kind": "successor-work",
              "url": "https://huggingface.co/swiss-ai/Apertus-v1.5-70B",
              "evidence": "The released models are the result of continued pretraining of Apertus 1.0, adding a multimodal mix of 4T tokens to the 8B model and 2T tokens to the 70B model."
            }
          ],
          "data_description": "15T tokens from 1800+ languages, filtered for toxicity and compliance.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "Switzerland"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Safety alignment",
              "Misuse risk assessment",
              "Constitutional AI"
            ],
            "Toxicity & harassment": [
              "Model-generated toxicity",
              "Implicit hate speech"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Misuse risk assessment",
              "Safety alignment"
            ],
            "Toxicity & harassment": [
              "Implicit hate speech",
              "Model-generated toxicity"
            ]
          },
          "tentative": false
        },
        {
          "title": "Apertus (Switzerland's fully open, compliant LLM)",
          "wid": "apertus-switzerland-s-fully-open-compliant-llm",
          "type": "project",
          "year": "2024-ongoing",
          "venue": "Swiss AI Initiative (EPFL + ETH Zurich + CSCS)",
          "link": "https://www.apertus-ai.org/",
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_topics": [
            "Apertus",
            "Multimodal content filtering",
            "Swiss AI Initiative",
            "Training data curation",
            "LLM safety",
            "Open-source LLM",
            "Data compliance",
            "Multilingual",
            "Digital sovereignty"
          ],
          "stage": "Prevention",
          "relevance": 3,
          "about": "Switzerland's flagship open large language model and the headline release of the Swiss AI Initiative, described as the largest open-science effort for AI foundation models worldwide. It is an institution-wide undertaking drawing in several EPFL groups alongside ETH Zurich and the CSCS centre, which runs training on the Alps supercomputer. The model trains only on openly available, opt-out-respecting data with toxic and personally identifiable content filtered out, and the initiative commits to publicly releasing transparent software, models and data.",
          "why": "It is a sovereign, content-safety-filtered open model that serves as a trustworthy substrate for MDH work.",
          "data": "Text, multilingual web-scale corpora",
          "lab": [
            "NLP",
            "MLO",
            "SDSC"
          ],
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Misuse risk assessment",
            "Safety alignment",
            "Training-corpus design"
          ],
          "key_terms": [
            "Data compliance",
            "Swiss AI Initiative",
            "Content safety filtering",
            "Fully open LLMs",
            "Low-resource languages"
          ],
          "models": [
            "Apertus-70B-Instruct"
          ],
          "method_qualifiers": [
            "Jailbreaking",
            "Constitutional AI",
            "Selective token masking"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Apertus toxicity classifiers",
              "kind": "tool"
            },
            {
              "name": "Apertus-70B-Instruct",
              "kind": "model"
            },
            {
              "name": "apertus-posttrain-romansh",
              "kind": "dataset"
            },
            {
              "name": "apertus-pretrain-poisonandcanaries",
              "kind": "dataset"
            },
            {
              "name": "apertus-pretrain-toxicity",
              "kind": "model"
            },
            {
              "name": "Swiss AI Charter",
              "kind": "framework"
            },
            {
              "name": "SwitzerlandQA",
              "kind": "benchmark"
            },
            {
              "name": "Apertus",
              "kind": "model",
              "url": "https://huggingface.co/swiss-ai",
              "evidence": "Hugging Face organisation 'swiss-ai' (Swiss AI Initiative) hosts the Apertus family, including swiss-ai/Apertus-8B-Instruct-2509, swiss-ai/Apertus-70B-Instruct-2509, swiss-ai/Apertus-70B-2509, swiss-ai/Apertus-v1.5-8B and swiss-ai/Apertus-v1.5-70B. The 8B-Instruct model card states: 'Apertus is trained while respecting opt-out consent of data owners (even retrospectively), and avoiding memorization of training data', released under Apache 2.0. ETH Zurich press release: 'Researchers from EPFL, ETH Zurich and CSCS have developed the large language model Apertus'."
            }
          ],
          "follow_up": [
            {
              "what": "Apertus v1.5, the multimodal successor generation (image and audio input, improved reasoning, tool use), released on Hugging Face",
              "kind": "model",
              "url": "https://huggingface.co/swiss-ai/Apertus-v1.5-70B",
              "evidence": "EPFL news 'Apertus 1.5: Building the next generation of open AI infrastructure': the update adds the ability to 'understand images and audio alongside text', with 'improved reasoning, stronger instruction following and better support for tool use'; 'The full model suite is available on Hugging Face'. The Hugging Face swiss-ai collection lists swiss-ai/Apertus-v1.5-70B (72B params) and swiss-ai/Apertus-v1.5-8B."
            },
            {
              "what": "Apertus technical report accepted at the ACL 2026 Main Conference: 'Apertus: Democratizing Open and Compliant LLMs for Global Language Environments'",
              "kind": "successor-work",
              "url": "https://arxiv.org/abs/2509.14233",
              "evidence": "apertus-ai.org: 'the technical report underlying the Apertus v1 large language model (LLM) has been accepted for presentation at the ACL 2026 Main Conference.' The report describes models 'pretrained exclusively on openly available data, retroactively respecting robots.txt exclusions and filtering for non-permissive, toxic, and personally identifiable content', trained on 15T tokens from over 1800 languages, released at 8B and 70B scales."
            },
            {
              "what": "Canton of Ticino deployment: in-house AI translation service for sensitive government documents",
              "kind": "deployment",
              "url": "https://actu.epfl.ch/news/apertus-15-building-the-next-generation-of-open--2/",
              "evidence": "In the Canton of Ticino, it powers an in-house AI translation service, allowing sensitive government documents to be translated without relying on commercial providers."
            },
            {
              "what": "MeditronFO, an EPFL framework for building medical large language models, uses Apertus as one of its foundation models",
              "kind": "successor-work",
              "url": "https://actu.epfl.ch/news/apertus-15-building-the-next-generation-of-open--2/",
              "evidence": "Researchers at EPFL have used Apertus as one of the foundation models for MeditronFO, the world's first fully open framework for building medical large language models."
            },
            {
              "what": "Bajour newsroom deployment: the Basel online news outlet runs Apertus locally to prepare local political news for its Basel Briefing newsletter",
              "kind": "deployment",
              "url": "https://actu.epfl.ch/news/apertus-15-building-the-next-generation-of-open--2/",
              "evidence": "The Basel-based online news outlet Bajour runs the locally hosted model in its newsroom to prepare local political news for its daily Basel Briefing newsletter."
            },
            {
              "what": "Distribution through Swisscom and the Public AI Inference Utility",
              "kind": "deployment",
              "url": "https://ethz.ch/en/news-and-events/eth-news/news/2025/09/press-release-apertus-a-fully-open-transparent-multilingual-language-model.html",
              "evidence": "AI researchers, professionals, and experienced enthusiasts can either access the model through the strategic partner Swisscom or download it from Hugging Face, and the Public AI Inference Utility provides access for people outside Switzerland."
            }
          ],
          "data_description": "Multilingual web-scale text corpora for training an open LLM.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "Switzerland"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Training-corpus design",
              "Safety alignment",
              "Misuse risk assessment"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Misuse risk assessment",
              "Safety alignment",
              "Training-corpus design"
            ]
          },
          "tentative": true
        },
        {
          "title": "Synthetic Disinformation Attacks on Automated Fact Verification Systems",
          "wid": "synthetic-disinformation-attacks-on-automated-fact-veri",
          "type": "publication",
          "year": 2022,
          "venue": "AAAI 2022 (Proceedings of the AAAI Conference on Artificial Intelligence, 36(10), 10581-10589)",
          "link": "https://ojs.aaai.org/index.php/AAAI/article/view/21302",
          "authors": [
            "Yibing Du (Stanford)",
            "Antoine Bosselut (EPFL)",
            "Christopher D. Manning (Stanford)"
          ],
          "epfl_authors": [
            "Antoine Bosselut (EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Fact-checking",
            "Adversarial attacks",
            "LLM-generated disinformation",
            "Evidence retrieval",
            "NLP"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "Study of whether automated fact-checking pipelines can be fooled by poisoning their evidence with machine-generated disinformation. It defines two attacks: adversarial addition, which injects generated false documents into the evidence repository, and adversarial modification, which rewrites existing evidence by swapping key terms or paraphrasing, and runs them against several fact-checking models on three benchmarks. The attacks cause large accuracy drops, and even a single injected sentence does major damage, while a control that swaps in random real sentences barely hurts, showing the harm comes from the content of the disinformation rather than from crowding out correct evidence.",
          "why": "It shows that LLM-generated disinformation can severely degrade the fact-verification systems built to counter misinformation.",
          "data": "Text (FEVER, SciFact, CovidFact)",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Detector robustness",
            "Evidence repository poisoning"
          ],
          "key_terms": [
            "Automated fact-checking",
            "Evidence poisoning",
            "Data voids",
            "Adversarial attacks",
            "GROVER"
          ],
          "models": [
            "BERT",
            "CorefBERT",
            "GROVER",
            "KGAT",
            "PEGASUS",
            "RoBERTa"
          ],
          "method_qualifiers": [
            "Data poisoning",
            "Adversarial evasion"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Adversarial disinformation dataset (ADVADD/ADVMOD)",
              "kind": "dataset"
            },
            {
              "name": "adversarial-factcheck",
              "kind": "dataset"
            },
            {
              "name": "ADVADD / ADVMOD adversarial evidence documents",
              "kind": "dataset",
              "url": "https://github.com/Yibing-Du/adversarial-factcheck",
              "evidence": "Paper, Reproducibility Checklist: \"We also introduce our own datasets of adversarial evidence generated by GROVER and PEGASUS (Zhang et al. 2019). They will be made publicly available with a license that allows for research use. For computational experiments in this paper, the main source code is available at: https://github.com/Yibing-Du/adversarial-factcheck\". Repository README: \"This repository contains code to reproduce main results in AAAI-22 paper 'Synthetic Disinformation Attacks on Automated Fact Verification Systems'\", citing du2022synthetic (Yibing Du, Antoine Bosselut, Christopher D. Manning). The adversarial-addition evidence is shipped in the repo tree as fever/data/advadd_full.json, fever/data/advadd_min.json, scifact/data/corpus_advadd.jsonl and covidfact/data/test1_advadd.tsv."
            }
          ],
          "data_description": "FEVER, SciFact and CovidFact benchmarks plus 74M crawled news documents.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Wikipedia"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Evidence repository poisoning",
              "Detector robustness"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "Detector robustness",
              "Evidence repository poisoning"
            ]
          },
          "tentative": false
        },
        {
          "title": "\"Flex Tape Can't Fix That\": Bias and Misinformation in Edited Language Models",
          "wid": "flex-tape-can-t-fix-that-bias-and-misinformation-in-edi",
          "type": "publication",
          "year": 2024,
          "venue": "EMNLP 2024 (Main Conference)",
          "link": "https://aclanthology.org/2024.emnlp-main.494/",
          "authors": [
            "Karina Halevy (EPFL / Carnegie Mellon)",
            "Anna Sotnikova (EPFL / U Maryland)",
            "Badr AlKhamissi (EPFL NLP)",
            "Syrielle Montariol (EPFL NLP)",
            "Antoine Bosselut (EPFL NLP)"
          ],
          "epfl_authors": [
            "Karina Halevy (EPFL / Carnegie Mellon)",
            "Anna Sotnikova (EPFL / U Maryland)",
            "Badr AlKhamissi (EPFL NLP)",
            "Syrielle Montariol (EPFL NLP)",
            "Antoine Bosselut (EPFL NLP)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "H"
          ],
          "mdh_topics": [
            "Model editing",
            "LLM bias",
            "Misinformation",
            "Sexism",
            "Xenophobia",
            "Fairness"
          ],
          "stage": "Monitoring",
          "relevance": 4,
          "about": "Study of a hidden cost of editing facts directly into a language model's weights: the edits leak into unrelated knowledge and amplify demographic bias. The authors build a benchmark of knowledge edits across properties such as gender, citizenship and birthplace, compare three editing methods across five models, and have annotators score the open-ended generations. All three methods amplify bias, with especially large confidence drops for Asian, African and Middle Eastern subjects and significant rises in sexism after gender edits and in xenophobia and racism after citizenship edits. Weight-based editing can pass standard specificity tests yet still inject misinformation and worsen bias against marginalised groups.",
          "why": "It shows that editing LLMs injects misinformation and amplifies sexism, xenophobia and racism against marginalised groups.",
          "data": "Text (SeeSaw-CF benchmark, 3516 edits, 734620 cloze prompts, 27010 open-ended prompts)",
          "themes": [
            "AI safety",
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Bias amplification",
            "Factual robustness",
            "Model-generated toxicity",
            "Safety alignment",
            "Sexism",
            "Xenophobia"
          ],
          "key_terms": [
            "Model editing",
            "Bias amplification",
            "Demographic bias",
            "SEESAW-CF",
            "LLM safety"
          ],
          "models": [
            "GPT-3.5",
            "GPT-J",
            "Llama 2",
            "Llama2",
            "Mistral"
          ],
          "method_qualifiers": [
            "MEMIT",
            "Model editing",
            "Manual annotation",
            "Cloze-completion probing"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "SEESAW-CF",
              "kind": "benchmark",
              "url": "https://github.com/ENSCMA2/flextape",
              "evidence": "Paper: 'We release our code and data publicly. 3' with footnote '3 https://github.com/ENSCMA2/flextape'. The repository README states: 'Here is the code used for the paper _\"Flex Tape Can't Fix That\": Pitfalls of Model Editing_, accepted to EMNLP 2024 (preprint https://arxiv.org/abs/2403.00180).' Its data/ directory holds the benchmark itself as seesaw_cf_P101.json, seesaw_cf_P103.json, seesaw_cf_P19_P101.json, seesaw_cf_P21_P101.json, seesaw_cf_P27_P101.json and further seesaw_cf_* files, matching the README's instruction that SEESAW-CF files 'begin with seesaw_cf_'."
            }
          ],
          "data_description": "3516 edits, 734k cloze prompts, 27k open-ended prompts across 5 LLMs.",
          "follow_up_checked": "2026-09-02",
          "targeted_group": [
            "African people",
            "Black people",
            "East Asian people",
            "Jewish people",
            "Middle Eastern people",
            "Transgender women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Bias amplification",
              "Factual robustness",
              "Safety alignment"
            ],
            "Toxicity & harassment": [
              "Model-generated toxicity",
              "Sexism",
              "Xenophobia"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Bias amplification",
              "Factual robustness",
              "Safety alignment"
            ],
            "Toxicity & harassment": [
              "Identity-targeted hate",
              "Model-generated toxicity"
            ]
          },
          "tentative": false
        },
        {
          "title": "A Logical Fallacy-Informed Framework for Argument Generation",
          "wid": "a-logical-fallacy-informed-framework-for-argument-gener",
          "type": "publication",
          "year": 2025,
          "venue": "NAACL 2025 (Long Papers)",
          "link": "https://aclanthology.org/2025.naacl-long.374/",
          "authors": [
            "Luca Mouchel",
            "Debjit Paul",
            "Shaobo Cui",
            "Robert West",
            "Antoine Bosselut",
            "Boi Faltings"
          ],
          "epfl_authors": [
            "Luca Mouchel",
            "Debjit Paul",
            "Shaobo Cui",
            "Robert West",
            "Antoine Bosselut",
            "Boi Faltings"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "Logical fallacies",
            "Argument generation",
            "LLM alignment",
            "Preference optimization",
            "Misinformation prevention"
          ],
          "stage": "Prevention",
          "relevance": 2,
          "about": "Study treating the tendency of LLMs to produce arguments with logical fallacies as a misinformation risk, since flawed reasoning is harder for readers to catch than an outright false statement. It proposes a training method, FIPO, that adds a fallacy classification objective across 13 fallacy categories and weights the penalty by how common each fallacy is in real data, lowering fallacy rates to 17 percent for Llama-2 and 19.5 percent for Mistral and largely fixing the most stubborn category, faulty generalization. The authors flag the framework as dual use, since the same approach could generate more coherent but deliberately deceptive arguments.",
          "why": "It reduces fallacious LLM-generated arguments the authors link to spreading misinformation, while flagging a dual-use risk.",
          "data": "Text (ExplaGraphs; 7872 fallacious arguments generated across 13 fallacy categories)",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Fallacy-aware alignment"
          ],
          "key_terms": [
            "Logical fallacies",
            "Argument generation",
            "LLM alignment",
            "Preference optimization",
            "AI-generated misinformation risk"
          ],
          "models": [
            "ChatGPT",
            "ELECTRA",
            "GPT-4",
            "Llama-2",
            "Mistral"
          ],
          "method_qualifiers": [
            "LoRA",
            "LLM-as-a-judge",
            "Preference optimisation",
            "DPO"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Fallacy-augmented preference dataset",
              "kind": "dataset"
            },
            {
              "name": "FIPO (Fallacy-Informed Preference Optimization)",
              "kind": "framework"
            },
            {
              "name": "Fallacy preference dataset",
              "kind": "dataset",
              "url": "https://github.com/lucamouchel/Logical-Fallacies/tree/main/data",
              "evidence": "Paper footnote 1, page 1: \"Our code and datasets are publicly available for research purposes at github.com/lucamouchel/Logical-Fallacies\". The repository's data/ directory (opened) contains the folders argumentation/, generated/, preference-data/, sft/, sft_rag/ and the file test_debate.txt, matching the paper's pipeline (ExplaGraphs argumentation data plus ChatGPT-generated fallacious counterparts forming the preference pairs)."
            },
            {
              "name": "FIPO",
              "kind": "framework",
              "url": "https://github.com/lucamouchel/Logical-Fallacies",
              "evidence": "GitHub repository description reads \"A Logical Fallacy-Informed Framework for Argument Generation\"; the README credits Luca Mouchel, Debjit Paul, Shaobo Cui, Robert West, Antoine Bosselut and Boi Faltings, NAACL 2025, and the src/ tree carries the data collection, supervised fine-tuning, preference optimization (DPO, KTO, PPO, CPO) and FIPO stages described in the paper."
            }
          ],
          "follow_up": [
            {
              "what": "Official code and data release for the paper (GitHub: lucamouchel/Logical-Fallacies)",
              "kind": "repository",
              "url": "https://github.com/lucamouchel/Logical-Fallacies",
              "evidence": "Paper footnote 1: \"Our code and datasets are publicly available for research purposes at github.com/lucamouchel/Logical-Fallacies\" - confirmed by opening the repository, whose description is the paper title and which implements the FIPO training and evaluation pipeline."
            }
          ],
          "data_description": "7872 generated fallacious arguments over ExplaGraphs topics and stances.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "United States"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Fallacy-aware alignment"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Fallacy-aware alignment"
            ]
          },
          "tentative": false
        },
        {
          "title": "Buy versus Build an LLM: A Decision Framework for Governments",
          "wid": "buy-versus-build-an-llm-a-decision-framework-for-govern",
          "type": "publication",
          "year": 2026,
          "venue": "ACM Technology Policy Council white paper (arXiv preprint)",
          "link": "https://arxiv.org/abs/2602.13033",
          "authors": [
            "Jiahao Lu",
            "Ziwei Xu",
            "William Tjhi",
            "Junnan Li",
            "Antoine Bosselut",
            "Pang Wei Koh",
            "Mohan Kankanhalli"
          ],
          "epfl_authors": [
            "Antoine Bosselut"
          ],
          "about": "Policy framework for governments deciding whether to buy access to existing commercial models, build domestic capabilities, or adopt hybrid approaches, at a time when LLM outputs are increasingly treated as trusted inputs to public decision-making and public discourse. It weighs these options across sovereignty, security, cost, economic and regional development, resource capability and sustainability, national context fit and the evolving cost-capability landscape, noting that such systems decide how issues are framed and what is treated as a reasonable or factual conclusion. Drawing on the development of Singapore's SEA-LION and Switzerland's Apertus, it adds practical lessons and does not prescribe a universal answer.",
          "themes": [
            "AI safety",
            "Platform governance & regulation"
          ],
          "subtopics": [
            "Misuse risk assessment",
            "Red-teaming",
            "Regulatory oversight & audit",
            "Safety alignment",
            "Technical standards",
            "Vendor lock-in"
          ],
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "NA",
          "key_terms": [
            "Digital sovereignty",
            "AI governance",
            "AI safety",
            "AI Sovereignty",
            "Buy vs build decision"
          ],
          "models": [
            "Apertus",
            "SEA-LION"
          ],
          "method_qualifiers": [],
          "events_cases": [
            "Apertus project",
            "SEA-LION",
            "SEA-LION development"
          ],
          "built_at_epfl": [],
          "data_description": "No empirical data (policy analysis).",
          "region_country": [
            "Singapore",
            "Switzerland"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Safety alignment",
              "Misuse risk assessment",
              "Red-teaming"
            ],
            "Platform governance & regulation": [
              "Technical standards",
              "Regulatory oversight & audit",
              "Vendor lock-in"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Misuse risk assessment",
              "Safety alignment"
            ],
            "Platform governance & regulation": [
              "Regulatory oversight & audit",
              "Technical standards"
            ]
          },
          "tentative": false
        },
        {
          "title": "Helpful to a Fault: Measuring Illicit Assistance in Multi-Turn, Multilingual LLM Agents",
          "wid": "helpful-to-a-fault-measuring-illicit-assistance-in-mult",
          "type": "publication",
          "year": 2026,
          "venue": "ICML 2026 (PMLR 306)",
          "link": "https://arxiv.org/abs/2602.16346",
          "authors": [
            "Nivya Talokar",
            "Ayush K Tarun",
            "Murari Mandal",
            "Maksym Andriushchenko",
            "Antoine Bosselut"
          ],
          "epfl_authors": [
            "Ayush K Tarun",
            "Antoine Bosselut"
          ],
          "about": "Paper introducing STING, an automated red-teaming framework that builds a step-by-step illicit plan under a benign persona and probes a target LLM agent over multiple turns with adaptive follow-ups, using judge agents to track phase completion. On AgentHarm scenarios it yields substantially higher illicit-task completion than single-turn prompting and chat-oriented multi-turn baselines, up to 107.1 percent higher than single-prompt instructions. Across six non-English languages, attack success and illicit-task completion do not consistently increase in lower-resource languages, unlike common chatbot findings. A simple safety system prompt cut harmful-task completion more than prompt filtering, with modest impact on benign tool use.",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Adaptive attacks",
            "Misuse risk assessment",
            "Safety alignment"
          ],
          "mdh_focus": [
            "D",
            "H"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "Prevention",
          "key_terms": [
            "Automated red-teaming",
            "AgentHarm benchmark",
            "LLM agents",
            "Agent safety",
            "Jailbreaking"
          ],
          "models": [
            "Claude Sonnet 4.5",
            "DeepSeek-V3.2",
            "GPT-5.1",
            "Gemini 3 Flash",
            "Llama Prompt Guard 2",
            "Qwen3-Next-80B-A3B-Instruct"
          ],
          "method_qualifiers": [
            "Jailbreaking",
            "Cox proportional hazards",
            "Automated red-teaming",
            "Survival analysis"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "STING",
              "kind": "framework"
            }
          ],
          "data_description": "44 AgentHarm behaviors, 176 prompts, red-teamed across 7 languages.",
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Adaptive attacks",
              "Misuse risk assessment",
              "Safety alignment"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Adaptive attacks",
              "Misuse risk assessment",
              "Safety alignment"
            ]
          },
          "tentative": false
        },
        {
          "title": "Tracking the Limits of Knowledge Propagation: How LLMs Fail at Multi-Step Reasoning with Conflicting Knowledge",
          "wid": "tracking-the-limits-of-knowledge-propagation-how-llms-f",
          "type": "publication",
          "year": 2026,
          "venue": "EACL 2026 (Long Papers)",
          "link": "https://aclanthology.org/2026.eacl-long.273/",
          "authors": [
            "Yiyang Feng",
            "Zeming Chen",
            "Haotian Wu",
            "Jiawei Zhou",
            "Antoine Bosselut"
          ],
          "epfl_authors": [
            "Yiyang Feng",
            "Zeming Chen",
            "Haotian Wu",
            "Antoine Bosselut"
          ],
          "about": "Benchmark, TRACK (Testing Reasoning Amid Conflicting Knowledge), for studying how LLMs propagate new knowledge through multi-step reasoning when it conflicts with the model's initial parametric knowledge. Its 1500 examples span multi-hop QA on recent Wikidata (WIKI), code generation with external APIs (CODE) and multi-step mathematical reasoning (MATH); each model is first probed for its knowledge gaps, then given the updated facts as conflicting knowledge for a downstream reasoning problem. On Llama-3.2, Qwen-3, GPT-4.1-mini and o4-mini, updated facts can worsen performance compared to providing none, some models degrade as more are provided, and failures stem from both an inability to faithfully integrate facts and flawed reasoning even when they are integrated.",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Capability evaluation",
            "Factual robustness",
            "Reasoning faithfulness"
          ],
          "mdh_focus": [
            "M"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "NA",
          "key_terms": [
            "Knowledge conflicts",
            "Knowledge propagation",
            "Multi-step reasoning",
            "Knowledge editing",
            "Knowledge conflict"
          ],
          "models": [
            "GPT-4.1-mini",
            "Llama-3.2",
            "Qwen-3",
            "o4-mini"
          ],
          "method_qualifiers": [
            "LLM-as-a-judge",
            "Fine-tuning",
            "Knowledge editing",
            "In-context learning"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "TRACK",
              "kind": "benchmark"
            }
          ],
          "data_description": "1500 examples across Wiki, Code, and Math scenarios.",
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Factual robustness",
              "Capability evaluation",
              "Reasoning faithfulness"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Factual robustness",
              "Misuse risk assessment"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "bunne",
      "name": "Charlotte Bunne",
      "url": "https://people.epfl.ch/charlotte.bunne",
      "unit": "AIMM",
      "faculty": "IC / SV",
      "mdh_focus": [],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "ML",
        "Healthcare AI",
        "Optimal Transport"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "jakob",
      "name": "Wenzel Jakob",
      "url": "https://people.epfl.ch/wenzel.jakob",
      "unit": "RGL",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [
        "Image"
      ],
      "techTypes": [
        "Rendering",
        "Computer Graphics",
        "Differentiable Rendering"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "fua",
      "name": "Pascal Fua",
      "url": "https://people.epfl.ch/pascal.fua",
      "unit": "CVLab",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [
        "Image",
        "Video"
      ],
      "techTypes": [
        "Computer Vision",
        "ML",
        "3D Reconstruction"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "chappelier",
      "name": "Jean-Cédric Chappelier",
      "url": "https://people.epfl.ch/jean-cedric.chappelier",
      "unit": "NLP",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "NLP",
        "Language Modeling"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "ford",
      "name": "Bryan Alexander Ford",
      "url": "https://people.epfl.ch/bryan.ford",
      "unit": "DEDIS",
      "faculty": "IC",
      "mdh_focus": [
        "M",
        "D",
        "H"
      ],
      "dataTypes": [
        "Data-agnostic"
      ],
      "techTypes": [
        "Proof of Personhood",
        "Digital Identity",
        "Sybil Detection",
        "Decentralised Systems",
        "Privacy",
        "Blockchain"
      ],
      "stage": "Prevention",
      "publications": [
        {
          "title": "Identity and Personhood in Digital Democracy: Evaluating Inclusion, Equality, Security, and Privacy in Pseudonym Parties and Other Proofs of Personhood",
          "wid": "identity-and-personhood-in-digital-democracy-evaluating",
          "type": "publication",
          "year": 2020,
          "venue": "arXiv technical report (EPFL Infoscience)",
          "link": "https://arxiv.org/abs/2011.02412",
          "authors": [
            "Bryan Ford"
          ],
          "epfl_authors": [
            "Bryan Ford"
          ],
          "mdh_relevance": "infrastructure",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Proof of personhood",
            "Digital identity",
            "Sybil attacks",
            "Online voting",
            "Content moderation governance"
          ],
          "stage": "Prevention",
          "relevance": 2,
          "about": "Technical report arguing that digital identity is neither necessary nor sufficient for digital democracy and proposing digital personhood as the missing foundation. It sets out four goals any proof-of-personhood mechanism should meet (inclusion, equality, security against Sybil attacks and social bots, and privacy) and proposes federated pseudonym parties: periodic in-person events where each attendee receives one anonymous cryptographic token per cycle. It evaluates alternatives such as government ID, biometrics, self-sovereign identity, proof-of-work, proof-of-stake and social-trust-network Sybil detection, and concludes that none satisfies all four goals at once while pseudonym parties plausibly can.",
          "why": "It proposes proof of personhood as a foundation against fake-account-driven disinformation, bot armies and Sybil attacks on online discourse.",
          "data": "N/A (theoretical)",
          "themes": [
            "NA"
          ],
          "subtopics": [],
          "key_terms": [
            "Proof of personhood",
            "Pseudonym parties",
            "Digital democracy",
            "Coercion resistance",
            "Sybil resistance"
          ],
          "models": [],
          "method_qualifiers": [
            "Threat modeling",
            "Sybil resistance"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "follow_up": [
            {
              "what": "TRIP / Votegral, the coercion-resistant in-person voter registration system from Ford's own DEDIS lab at EPFL, which implements the privacy-booth kiosk printing one real and several indistinguishable fake tokens that this paper proposed, and which cites this paper. Published at SOSP 2025; prototype code at github.com/dedis/votegral.",
              "kind": "successor-work",
              "url": "https://bford.info/pub/sec/trip/",
              "evidence": "TRIP's bibliography entry, verbatim: \"Ford (2020a) Bryan Ford. 2020a. Identity and Personhood in Digital Democracy: Evaluating Inclusion, Equality, Security, and Privacy in Pseudonym Parties and Other Proofs of Personhood. https://doi.org/10.48550/arXiv.2011.02412\" - and the in-text use of it, verbatim: \"Integrating TRIP's coercion-resistance mechanism into in-person pseudonym parties (Ford and Strauss, 2008) as a proof-of-personhood protocol (Borge et al., 2017; Ford, 2020a; Siddarth et al., 2020a), in particular, could help address this challenge and enable truly democratic DAOs and other democratic computing platforms in the future.\" The mechanism is the one this paper specified: this paper says \"the attendee then enters one of several curtained privacy booths. The attendee inserts the ticket into a kiosk in the privacy both, which prints one real token and several fake tokens on paper... Upon leaving the privacy booth, however, only the attendee knows which is the real token, and cannot subsequently prove which is which to anyone else.\" TRIP's abstract says \"Votegral's registration component, TRIP, gives voters a kiosk in a privacy booth with which to print real and fake credentials on paper, eliminating dependence on trusted hardware in credential issuance. The voter learns and can verify in the privacy booth which credential is real, but real and fake credentials thereafter appear indistinguishable to others.\" Bryan Ford is an author of TRIP."
            }
          ],
          "data_description": "No empirical data (conceptual design and security analysis).",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {},
          "theme_qualifiers_canonical": {},
          "tentative": false
        },
        {
          "title": "Spotlight on risk: Using 'proof of personhood' to tackle social media risks",
          "wid": "using-proof-of-personhood-to-tackle-social-media-risks",
          "type": "report",
          "year": 2021,
          "venue": "EPFL International Risk Governance Center (IRGC) Spotlight on Risk series",
          "link": "https://dhcenter-unil-epfl.ch/en/2021/03/26/epfl-irgc-spotlight-on-risk-using-proof-of-personhood-to-tackle-social-media-risks/",
          "authors": [
            "Aengus Collins (EPFL IRGC)",
            "Bryan Ford (DEDIS)"
          ],
          "epfl_authors": [
            "Aengus Collins (EPFL IRGC)",
            "Bryan Ford (DEDIS)"
          ],
          "mdh_relevance": "infrastructure",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "Proof of personhood",
            "Sockpuppets",
            "Bot armies",
            "Deepfakes",
            "Social media accountability",
            "Anonymity"
          ],
          "stage": "Prevention + Mitigation",
          "relevance": 3,
          "about": "Policy brief framing the core social-media problem as the cheap, replaceable and automatable nature of fake virtual identities, which amplifies bot and sockpuppet abuse and feeds the spread of misinformation, fake news and conspiracy theories. As a low-tech response it proposes pseudonym parties: simultaneous in-person events where each attendee receives one anonymous cryptographic token per cycle, attesting that a real person showed up without revealing identifying data. Suggested uses include blocking abusive tokens rather than accounts, producing verified per-person like and follow counts, and replacing CAPTCHAs. It flags operational challenges and free-speech tensions and recommends small voluntary pilots before any scaling.",
          "why": "It frames proof of personhood as a defence against fake-account-driven misinformation, bot armies and conspiracy theories.",
          "data": "N/A (policy analysis)",
          "lab": "IRGC",
          "themes": [
            "Influence operations & coordinated manipulation",
            "Platform governance & regulation"
          ],
          "subtopics": [
            "Online anonymity",
            "Pseudonym parties",
            "Regulatory oversight & audit"
          ],
          "key_terms": [
            "Proof of personhood",
            "Pseudonym parties",
            "Online anonymity",
            "Sockpuppets",
            "Fake accounts"
          ],
          "models": [],
          "method_qualifiers": [
            "Sybil resistance"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "No empirical data (policy analysis).",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "Platform governance & regulation": [
              "Online anonymity",
              "Pseudonym parties",
              "Regulatory oversight & audit"
            ]
          },
          "theme_qualifiers_canonical": {
            "Platform governance & regulation": [
              "Online anonymity",
              "Regulatory oversight & audit"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "jaggi",
      "name": "Martin Jaggi",
      "url": "https://people.epfl.ch/martin.jaggi",
      "unit": "MLO",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [
        "Text",
        "Data-agnostic"
      ],
      "techTypes": [
        "LLMs",
        "Data Governance",
        "Federated Learning",
        "LoRA",
        "On-device ML",
        "Optimisation"
      ],
      "stage": "Prevention",
      "publications": []
    },
    {
      "id": "argyraki",
      "name": "Katerina Argyraki",
      "url": "https://people.epfl.ch/katerina.argyraki",
      "unit": "NAL",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [
        "Data-agnostic"
      ],
      "techTypes": [
        "Computer Networks",
        "Network Security"
      ],
      "stage": "Mitigation",
      "publications": []
    },
    {
      "id": "thiran_jp",
      "name": "Jean-Philippe Thiran",
      "url": "https://people.epfl.ch/jean-philippe.thiran",
      "unit": "LTS5",
      "faculty": "STI / SV",
      "mdh_focus": [],
      "dataTypes": [
        "Image",
        "Video",
        "Audio"
      ],
      "techTypes": [
        "Signal Processing",
        "ML",
        "Biomedical Imaging"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "flammarion",
      "name": "Nicolas Flammarion",
      "url": "https://people.epfl.ch/nicolas.flammarion",
      "unit": "TML",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "LLMs",
        "LLM Safety",
        "Adversarial Robustness",
        "Jailbreak Evaluation",
        "ML",
        "Optimisation"
      ],
      "stage": "Monitoring",
      "publications": [
        {
          "title": "Jailbreaking Leading Safety-Aligned LLMs with Simple Adaptive Attacks",
          "wid": "jailbreaking-leading-safety-aligned-llms-with-simple-ad",
          "type": "publication",
          "year": 2025,
          "venue": "ICLR 2025 (International Conference on Learning Representations)",
          "link": "https://arxiv.org/abs/2404.02151",
          "authors": [
            "Maksym Andriushchenko",
            "Francesco Croce",
            "Nicolas Flammarion"
          ],
          "epfl_authors": [
            "Maksym Andriushchenko",
            "Francesco Croce",
            "Nicolas Flammarion"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_topics": [
            "LLM safety",
            "Jailbreak",
            "Adversarial robustness",
            "Hate-speech generation risk",
            "Misinformation generation risk"
          ],
          "stage": "Prevention",
          "relevance": 2,
          "about": "Study showing that even recent safety-aligned large language models can be reliably jailbroken with simple adaptive attacks. The method pairs a manually designed prompt template with a random search over an appended suffix that maximises the probability of a compliant target token, using transfer or prefilling variants for models that hide log-probabilities. Adapting the attack to each model's weaknesses, it reports a near-total attack success rate across many leading models, far exceeding prior results.",
          "why": "It shows alignment fails to block harmful generation, where the targeted content explicitly includes misinformation and hateful material.",
          "data": "Text, 50 harmful AdvBench requests",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Adaptive attacks",
            "Adversarial robustness",
            "Safety alignment"
          ],
          "key_terms": [
            "Jailbreaking",
            "Safety alignment",
            "Trojan detection",
            "Adaptive attacks",
            "Adversarial robustness"
          ],
          "models": [
            "Claude-3.5",
            "GPT-3.5",
            "GPT-4",
            "GPT-4o",
            "Gemma-7B",
            "Llama-3-Instruct-8B",
            "Mistral-7B",
            "Nemotron-4-340B",
            "Phi-3-Mini",
            "R2D2",
            "Vicuna-13B"
          ],
          "method_qualifiers": [
            "Jailbreaking",
            "Random search",
            "Self-transfer",
            "Black-box attack"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "llm-adaptive-attacks",
              "kind": "tool"
            }
          ],
          "follow_up": [
            {
              "what": "Public code and attack-log repository for the paper (tml-epfl/llm-adaptive-attacks), MIT-licensed, maintained through September 2024 with added runs for Claude Sonnet 3.5, Nemotron-4-340B, Phi-3-Mini and Llama-3-8B",
              "kind": "repository",
              "url": "https://github.com/tml-epfl/llm-adaptive-attacks",
              "evidence": "Paper abstract: \"For reproducibility purposes, we provide the code, logs, and jailbreak artifacts in the JailbreakBench format at https://github.com/tml-epfl/llm-adaptive-attacks .\" Repository README header: \"# Jailbreaking Leading Safety-Aligned LLMs with Simple Adaptive Attacks\n\n**Maksym Andriushchenko (EPFL), Francesco Croce (EPFL), Nicolas Flammarion (EPFL)**\n\n**Paper:** https://arxiv.org/abs/2404.02151\n\n**ICLR 2025**\"."
            },
            {
              "what": "Released jailbreak artifacts (successful adversarial prompts and generations for each targeted model) exported in the JailbreakBench artifact format",
              "kind": "dataset",
              "url": "https://github.com/tml-epfl/llm-adaptive-attacks/tree/main/jailbreak_artifacts",
              "evidence": "Repository README, Updates section: \"**4 May 2024**: We've exported most of our jailbreak artifacts (see the `jailbreak_artifacts` folder) in a convenient form following the [JailbreakBench](https://github.com/JailbreakBench/jailbreakbench) [format](https://github.com/JailbreakBench/artifacts).\" and \"**15 June 2024**: We've added jailbreak artifacts for Nemotron-4-340B-Instruct\". Folder URL confirmed to resolve (HTTP 200)."
            },
            {
              "what": "Winning submission code for the SaTML'24 Trojan Detection Competition, using the restricted-token random search described in this paper",
              "kind": "repository",
              "url": "https://github.com/fra31/rlhf-trojan-competition-submission",
              "evidence": "Repository README of this paper's own repo: \"## Trojan detection code\nFor the code used to obtain the 1st place in the SatML'24 Trojan Detection Competition, see [https://github.com/fra31/rlhf-trojan-competition-submission](https://github.com/fra31/rlhf-trojan-competition-submission).\" The paper's abstract states the same connection: \"we show how to use random search on a restricted set of tokens for finding trojan strings in poisoned models ... which is the algorithm that brought us the first place in the SaTML'24 Trojan Detection Competition.\" The linked repo's own README describes that method: \"Our method relies on random search (RS) in the token space to find the triggers, with the goal of minimizing the average reward on training points.\""
            }
          ],
          "data_description": "50 harmful requests from AdvBench, run against a dozen production LLMs.",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Adaptive attacks",
              "Adversarial robustness",
              "Safety alignment"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Adaptive attacks",
              "Adversarial robustness",
              "Safety alignment"
            ]
          },
          "tentative": false
        },
        {
          "title": "Competition Report: Finding Universal Jailbreak Backdoors in Aligned LLMs",
          "wid": "competition-report-finding-universal-jailbreak-backdoor",
          "type": "publication",
          "year": 2024,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2404.14461",
          "authors": [
            "Javier Rando",
            "Francesco Croce",
            "Kryštof Mitka",
            "Stepan Shabalin",
            "Maksym Andriushchenko",
            "Nicolas Flammarion",
            "Florian Tramèr"
          ],
          "epfl_authors": [
            "Francesco Croce",
            "Maksym Andriushchenko",
            "Nicolas Flammarion"
          ],
          "about": "Report on a competition to find universal jailbreak backdoors in aligned language models. Models aligned to prevent harmful content like misinformation can be poisoned through their safety training data with a backdoor string that enables harmful responses when added to any prompt. The organizers poisoned 5 instances of LLaMA-2 (7B), each with a different backdoor and a 25 percent poisoning rate, and scored submissions with a reward model. Across 12 valid submissions, only one trojan, for one model, elicited worse responses than the injected backdoors; the two best teams narrowed the search using embedding differences across models. The report releases the first suite of universally backdoored models and datasets.",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Adaptive attacks",
            "Adversarial robustness",
            "RLHF poisoning"
          ],
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "Prevention",
          "key_terms": [
            "Universal jailbreak backdoors",
            "RLHF poisoning",
            "LLM alignment",
            "AI safety",
            "Alignment robustness"
          ],
          "models": [
            "LLaMA-2 (7B)"
          ],
          "method_qualifiers": [
            "Data poisoning",
            "Jailbreaking",
            "RLHF",
            "Adversarial evasion"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Backdoor detection datasets",
              "kind": "dataset"
            },
            {
              "name": "RLHF Trojan Competition dataset splits",
              "kind": "dataset"
            },
            {
              "name": "RLHF Trojan Competition suite of poisoned LLaMA-2 models",
              "kind": "model"
            },
            {
              "name": "suite of universally backdoored LLMs",
              "kind": "model"
            },
            {
              "name": "Universally backdoored LLaMA-2 models",
              "kind": "model"
            }
          ],
          "data_description": "42000 training entries, 500 validation, 2300 test from Anthropic harmless dataset.",
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Adversarial robustness",
              "RLHF poisoning",
              "Adaptive attacks"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Adaptive attacks",
              "Adversarial robustness"
            ]
          },
          "tentative": false
        },
        {
          "title": "Does Refusal Training in LLMs Generalize to the Past Tense?",
          "wid": "does-refusal-training-in-llms-generalize-to-the-past-te",
          "type": "publication",
          "year": 2025,
          "venue": "ICLR 2025",
          "link": "https://arxiv.org/abs/2407.11969",
          "authors": [
            "Maksym Andriushchenko",
            "Nicolas Flammarion"
          ],
          "epfl_authors": [
            "Maksym Andriushchenko",
            "Nicolas Flammarion"
          ],
          "about": "Study revealing a generalization gap in LLM refusal training: simply reformulating a harmful request in the past tense is often sufficient to jailbreak many state-of-the-art models. Using GPT-3.5 Turbo to produce 20 reformulation attempts for 100 harmful requests from JBB-Behaviors, the attack success rate on GPT-4o rose from 1 percent with direct requests to 88 percent according to a GPT-4 judge, while future-tense reformulations were less effective. The attack succeeded less often on harassment, disinformation and sexual or adult content, and fine-tuning GPT-3.5 Turbo with refusals to past-tense reformulations can cut the attack success rate to 0 percent, although overrefusals must be carefully controlled.",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Adaptive attacks",
            "Adversarial robustness",
            "Safety alignment"
          ],
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "Prevention",
          "key_terms": [
            "Past-tense reformulation",
            "Refusal training",
            "Jailbreaking",
            "Alignment generalization",
            "JBB-Behaviors"
          ],
          "models": [
            "Claude-3.5 Sonnet",
            "GPT-3.5 Turbo",
            "GPT-4o",
            "GPT-4o mini",
            "GPT-4o-mini",
            "Gemma-2 9B",
            "Llama-3 8B",
            "Phi-3-Mini",
            "R2D2",
            "o1-mini",
            "o1-preview"
          ],
          "method_qualifiers": [
            "Jailbreaking",
            "Adversarial prompting",
            "LLM-as-a-judge",
            "Safety fine-tuning"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "llm-past-tense",
              "kind": "tool"
            }
          ],
          "data_description": "100 harmful behaviors from JBB-Behaviors across 10 harm categories, tested on 10 LLMs.",
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Adversarial robustness",
              "Adaptive attacks",
              "Safety alignment"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Adaptive attacks",
              "Adversarial robustness",
              "Safety alignment"
            ]
          },
          "tentative": false
        },
        {
          "title": "HalluHard: A Hard Multi-Turn Hallucination Benchmark",
          "wid": "halluhard-a-hard-multi-turn-hallucination-benchmark",
          "type": "publication",
          "year": 2026,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2602.01031",
          "authors": [
            "Dongyang Fan",
            "Sebastien Delsad",
            "Nicolas Flammarion",
            "Maksym Andriushchenko"
          ],
          "epfl_authors": [
            "Dongyang Fan",
            "Sebastien Delsad",
            "Nicolas Flammarion"
          ],
          "about": "Benchmark for hallucination in multi-turn, open-ended conversations with large language models, built from 950 seed questions across four high-stakes domains: legal cases, research questions, medical guidelines and coding. Models must support factual claims with inline citations, and a judging pipeline uses web search to fetch full-text sources, including PDFs, to check whether the cited material supports each claim. Across frontier proprietary and open-weight models, hallucinations remain substantial even with web search (about 30 percent for the strongest configuration, Opus-4.5 with web search), rise in later turns on citation-grounded tasks (not in coding), which the authors attribute to error propagation, and are far more often content-grounding failures than reference failures.",
          "themes": [
            "AI safety",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Claim credibility inference",
            "Content grounding",
            "Factual robustness",
            "LLM hallucination",
            "Multi-turn error propagation",
            "Source referencing practices"
          ],
          "mdh_focus": [
            "M"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "Monitoring",
          "key_terms": [
            "LLM hallucination",
            "Multi-turn dialogue",
            "Niche knowledge",
            "Citation grounding",
            "Claim verification"
          ],
          "models": [
            "Claude-Haiku-4.5",
            "Claude-Sonnet-4.5",
            "DeepSeek-V3.2",
            "GLM-4.7",
            "GPT-5",
            "GPT-5-mini",
            "GPT-5-nano",
            "GPT-5.2",
            "GPT-5.2-thinking",
            "Gemini-3-Flash",
            "Kimi-K2"
          ],
          "method_qualifiers": [
            "LLM-as-a-judge",
            "Algorithmic auditing",
            "Atomic claim decomposition"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "HalluHard",
              "kind": "benchmark"
            }
          ],
          "data_description": "950 seed questions across 4 domains; 17 frontier LLMs evaluated.",
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Factual robustness",
              "Multi-turn error propagation",
              "LLM hallucination"
            ],
            "Verification & content authenticity": [
              "Claim credibility inference",
              "Source referencing practices",
              "Content grounding"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Factual robustness"
            ],
            "Verification & content authenticity": [
              "Claim credibility inference",
              "Source referencing practices"
            ]
          },
          "tentative": false
        },
        {
          "title": "JailbreakBench: An Open Robustness Benchmark for Jailbreaking Large Language Models",
          "wid": "jailbreakbench-an-open-robustness-benchmark-for-jailbre",
          "type": "publication",
          "year": 2024,
          "venue": "NeurIPS 2024 (Datasets and Benchmarks Track)",
          "link": "https://arxiv.org/abs/2404.01318",
          "authors": [
            "Patrick Chao",
            "Edoardo Debenedetti",
            "Alexander Robey",
            "Maksym Andriushchenko",
            "Francesco Croce",
            "Vikash Sehwag",
            "Edgar Dobriban",
            "Nicolas Flammarion",
            "George J. Pappas",
            "Florian Tramèr",
            "Hamed Hassani",
            "Eric Wong"
          ],
          "epfl_authors": [
            "Maksym Andriushchenko",
            "Francesco Croce",
            "Nicolas Flammarion"
          ],
          "about": "Benchmark for jailbreak attacks, which cause large language models to generate harmful, unethical or otherwise objectionable content, and for defenses against them. It provides a repository of adversarial prompts, an evaluation framework, a leaderboard and the JBB-Behaviors dataset of 100 misuse behaviors in ten categories corresponding to OpenAI's usage policies, one of them disinformation, each matched with a benign behavior. Llama-3-70B is adopted as judge after a human evaluation of six classifiers. Tests of four attacks show that even recent and closed-source undefended models are highly vulnerable, with Prompt with RS reaching 90 percent attack success on Llama-2 and 78 percent on GPT-4.",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Adaptive attacks",
            "Adversarial robustness",
            "Safety alignment"
          ],
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 3,
          "stage": "Prevention + Monitoring",
          "key_terms": [
            "Red-teaming",
            "LLM-as-a-judge",
            "Jailbreaking",
            "LLM safety",
            "Adversarial prompts"
          ],
          "models": [
            "GPT-3.5",
            "GPT-3.5-Turbo-1106",
            "GPT-4",
            "GPT-4-0125-Preview",
            "Llama Guard 2",
            "Llama-2-7B-chat-hf",
            "Mixtral",
            "Vicuna-13B-v1.5"
          ],
          "method_qualifiers": [
            "Jailbreaking",
            "LLM-as-a-judge",
            "Adaptive attacks",
            "Adversarial evasion"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "JailbreakBench",
              "kind": "benchmark"
            },
            {
              "name": "jailbreakbench library",
              "kind": "tool"
            },
            {
              "name": "JBB-Behaviors",
              "kind": "dataset"
            }
          ],
          "data_description": "100 harmful and 100 benign behaviors; 300 labelled prompt-response pairs.",
          "targeted_group": [
            "Children"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Adaptive attacks",
              "Adversarial robustness",
              "Safety alignment"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Adaptive attacks",
              "Adversarial robustness",
              "Safety alignment"
            ]
          },
          "tentative": false
        },
        {
          "title": "On the Adversarial Robustness of Discrete Image Tokenizers",
          "wid": "on-the-adversarial-robustness-of-discrete-image-tokeniz",
          "type": "publication",
          "year": 2026,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2602.18252",
          "authors": [
            "Rishika Bhagwatkar",
            "Irina Rish",
            "Nicolas Flammarion",
            "Francesco Croce"
          ],
          "epfl_authors": [
            "Nicolas Flammarion"
          ],
          "about": "Study of the adversarial robustness of discrete image tokenizers, which encode images as tokens from a finite vocabulary for multimodal systems, presented as the first work on the topic. It formulates unsupervised attacks that perturb the features a tokenizer extracts, effective across classification, multimodal retrieval and captioning, and able to make a multimodal LLM output a target (malicious) caption without direct access to the LLM. As a defence, it fine-tunes tokenizers with unsupervised adversarial training, keeping other components frozen, improving robustness on unseen tasks and data; on UniTok-MLLM, targeted attacks with full model access, aimed at fraud, manipulation and harassment captions, succeeded with the original tokenizer, not the robust one.",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Adaptive attacks",
            "Adversarial robustness",
            "Multimodal safety"
          ],
          "mdh_focus": [
            "D",
            "H"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "Prevention",
          "key_terms": [
            "Discrete image tokenizers",
            "Adversarial robustness",
            "Unsupervised adversarial training",
            "Multimodal foundation models",
            "Adversarial training"
          ],
          "models": [
            "FlexTok",
            "FuseLIP",
            "LLaMA-2-7B",
            "TiTok-BL128",
            "UniTok-MLLM"
          ],
          "method_qualifiers": [
            "Adversarial attack"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "ImageNet-1k, CC3M, Imagenette, Caltech101, COCO images and VQA benchmarks.",
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Adversarial robustness",
              "Adaptive attacks",
              "Multimodal safety"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Adaptive attacks",
              "Adversarial robustness"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "miranda",
      "name": "Miranda Wei",
      "url": "https://people.epfl.ch/miranda.wei",
      "unit": "",
      "faculty": "IC (arriving autumn 2026)",
      "mdh_focus": [
        "H",
        "D"
      ],
      "dataTypes": [
        "Text",
        "Image"
      ],
      "techTypes": [
        "Social Computing",
        "Survey Methods",
        "Qualitative Interviews",
        "Online Harms Research",
        "Deepfakes"
      ],
      "stage": "Monitoring",
      "publications": [
        {
          "title": "\"Violation of my body:\" Perceptions of AI-generated non-consensual (intimate) imagery",
          "wid": "violation-of-my-body-perceptions-of-ai-generated-non-co",
          "type": "publication",
          "year": 2024,
          "venue": "USENIX SOUPS 2024 (Symposium on Usable Privacy and Security)",
          "link": "https://www.usenix.org/conference/soups2024/presentation/brigham",
          "authors": [
            "Natalie Grace Brigham (University of Washington)",
            "Miranda Wei (University of Washington)",
            "Tadayoshi Kohno (University of Washington)",
            "Elissa M. Redmiles (Georgetown)"
          ],
          "epfl_authors": [
            "Miranda Wei (University of Washington)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "H"
          ],
          "mdh_topics": [
            "AIG-NCII",
            "Deepfakes",
            "Image-based sexual abuse",
            "Gender-based violence",
            "Public attitudes"
          ],
          "stage": "NA",
          "relevance": 5,
          "about": "Vignette-based survey of public attitudes toward AI-generated non-consensual intimate imagery. It varies the behaviour (creating, sharing, resharing, or seeking out such content), the creator's relationship to the target, how explicit the depiction is, and the intent. Most respondents rated creation and all forms of sharing as unacceptable, while seeking out such content drew more mixed views, and acceptability shifted with the behaviour, creator, intent, and the respondent's gender and consent attitudes. The work frames this imagery as image-based sexual abuse and technology-facilitated gender-based violence.",
          "why": "It studies deepfake-based non-consensual intimate imagery as image-based sexual abuse and tech-facilitated gender-based violence.",
          "data": "Text, Survey (315 U.S. Prolific respondents; 2x3x3 vignette factorial design)",
          "themes": [
            "Gender-based violence & misogyny",
            "Online sexual abuse & image-based abuse"
          ],
          "subtopics": [
            "Consent norms",
            "Deepfake pornography",
            "Gendered attitude gaps",
            "Public acceptability"
          ],
          "key_terms": [
            "Image-based sexual abuse",
            "Sexual consent attitudes",
            "Deepfakes",
            "Public attitudes",
            "AIG-NCII"
          ],
          "models": [],
          "method_qualifiers": [
            "Thematic analysis",
            "Vignette-based survey",
            "Mixed-effects regression",
            "Randomized controlled trial"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "Survey of 315 US Prolific respondents, 861 open-text rationales.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "United States"
          ],
          "targeted_group": [
            "Marginalised genders",
            "Marginalized genders",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Gender-based violence & misogyny": [
              "Gendered attitude gaps"
            ],
            "Online sexual abuse & image-based abuse": [
              "Consent norms",
              "Deepfake pornography",
              "Public acceptability"
            ]
          },
          "theme_qualifiers_canonical": {
            "Gender-based violence & misogyny": [
              "Gendered attitude gaps"
            ],
            "Online sexual abuse & image-based abuse": [
              "Consent norms",
              "Deepfake pornography",
              "Public acceptability"
            ]
          },
          "tentative": false
        },
        {
          "title": "\"We're utterly ill-prepared to deal with something like this\": Teachers' Perspectives on Student Generation of Synthetic Nonconsensual Explicit Imagery",
          "wid": "we-re-utterly-ill-prepared-to-deal-with-something-like-",
          "type": "publication",
          "year": 2025,
          "venue": "CHI 2025 (ACM CHI Conference on Human Factors in Computing Systems)",
          "link": "https://dl.acm.org/doi/abs/10.1145/3706598.3713226",
          "authors": [
            "Miranda Wei",
            "Christina Yeung",
            "Franziska Roesner",
            "Tadayoshi Kohno"
          ],
          "epfl_authors": [
            "Miranda Wei"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "H"
          ],
          "mdh_topics": [
            "AIG-NCII",
            "K-12 schools",
            "Teacher training",
            "Tech-facilitated GBV"
          ],
          "stage": "Prevention + Mitigation",
          "relevance": 5,
          "about": "Qualitative interview study of how US K-12 teachers perceive and would handle student-on-student AI-generated nonconsensual explicit imagery. None of the teachers interviewed knew of an incident at their own school, but most expected it to grow, and they described an acute lack of institutional protocols, policy guidance and training to respond. They proposed interventions such as better reporting mechanisms, more emphasis on consent in sex education and updated technology policies, but disagreed on what consequences student creators should face. The work frames the issue as technology-facilitated gender-based violence at the school level.",
          "why": "It studies how schools confront synthetic nonconsensual explicit imagery as technology-facilitated gender-based violence.",
          "data": "Qualitative interviews with US K-12 teachers",
          "themes": [
            "Gender-based violence & misogyny",
            "Media literacy & public resilience",
            "Online sexual abuse & image-based abuse"
          ],
          "subtopics": [
            "Consent education",
            "Deepfake pornography",
            "Digital literacy",
            "Gendered victimisation",
            "Men's rights groups",
            "Slut-shaming"
          ],
          "key_terms": [
            "Deepfake nudes",
            "Image-based sexual abuse",
            "School policy",
            "Synthetic nonconsensual explicit imagery",
            "Consent education"
          ],
          "models": [],
          "method_qualifiers": [
            "Reflexive thematic analysis",
            "Semi-structured interviews"
          ],
          "events_cases": [
            "SNCEI incidents at US middle/high schools (2023-2024)",
            "Student-created synthetic nonconsensual explicit imagery in US middle and high schools"
          ],
          "built_at_epfl": [],
          "data_description": "Interviews with 17 US middle and high school teachers, 2024.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Instagram",
            "Snapchat",
            "TikTok"
          ],
          "region_country": [
            "United States"
          ],
          "targeted_group": [
            "Children",
            "Girls",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Gender-based violence & misogyny": [
              "Gendered victimisation",
              "Slut-shaming",
              "Men's rights groups"
            ],
            "Media literacy & public resilience": [
              "Consent education",
              "Digital literacy"
            ],
            "Online sexual abuse & image-based abuse": [
              "Deepfake pornography"
            ]
          },
          "theme_qualifiers_canonical": {
            "Gender-based violence & misogyny": [
              "Gendered victimisation",
              "Men's rights groups"
            ],
            "Media literacy & public resilience": [
              "Consent education",
              "Digital literacy",
              "Practitioner preparedness"
            ],
            "Online sexual abuse & image-based abuse": [
              "Consent norms",
              "Deepfake pornography",
              "Youth perpetration"
            ]
          },
          "tentative": false
        },
        {
          "title": "\"TikTok, Do Your Thing\": User Reactions to Social Surveillance in the Public Sphere",
          "wid": "-tiktok-do-your-thing-user-reactions-to-social-surveill",
          "type": "publication",
          "year": 2025,
          "link": "https://arxiv.org/abs/2506.20884",
          "authors": [
            "Meira Gilbert",
            "Miranda Wei",
            "Lindah Kotut"
          ],
          "about": "Miranda Wei's affiliation on this paper is the University of Washington, before her appointment at EPFL. Qualitative study of \"TikTok, Do Your Thing\", a viral trend in which users try to identify strangers they see in public through information crowd-sourcing, typically for romantic purposes. The authors analysed 60 TikTok videos and 1901 user comments, finding that 19 individuals were successfully identified and that supportive comments (n=883) were more than double the disapproving ones (n=310). Supportive comments showed genuine interest and empathy, while disapproving comments raised concerns about inappropriate relationships, stalking, consent and gendered double standards. The findings are discussed in relation to the normalization of interpersonal surveillance and online stalking, and as an evolution of social surveillance.",
          "themes": [
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Consent norms",
            "Doxxing risk",
            "Online stalking"
          ],
          "mdh_focus": [
            "H"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "Monitoring",
          "key_terms": [
            "Social surveillance",
            "Online stalking",
            "Contextual integrity",
            "TikTok",
            "Consent"
          ],
          "models": [],
          "method_qualifiers": [
            "Reflexive thematic analysis"
          ],
          "events_cases": [
            "TikTok, Do Your Thing trend"
          ],
          "built_at_epfl": [],
          "data_description": "60 TikTok videos, 1901 comments, Oct 2024-Jan 2025.",
          "platform": [
            "TikTok"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Toxicity & harassment": [
              "Online stalking",
              "Doxxing risk",
              "Consent norms"
            ]
          },
          "theme_qualifiers_canonical": {},
          "tentative": false
        },
        {
          "title": "Understanding Help-Seeking and Help-Giving on Social Media for Image-Based Sexual Abuse",
          "wid": "understanding-help-seeking-and-help-giving-on-social-me",
          "type": "publication",
          "year": 2024,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2406.12161",
          "authors": [
            "Miranda Wei",
            "Sunny Consolvo",
            "Patrick Gage Kelley",
            "Tadayoshi Kohno",
            "Tara Matthews",
            "Sarah Meiklejohn",
            "Franziska Roesner",
            "Renee Shelby",
            "Kurt Thomas",
            "Rebecca Umbach"
          ],
          "about": "Miranda Wei's affiliations on this paper are the University of Washington and Google, before her appointment at EPFL. Study of how adults seek and receive help on Reddit for image-based sexual abuse (IBSA) across seven types: financial and nonfinancial sextortion, nonconsensual synthetic explicit imagery, pressurized sexting, cyberflashing, nonconsensual explicit imagery and recorded sexual assault. An LLM pipeline sifted 5.7 million English-language posts from relationship and advice subcommunities, flagging 113K likely about IBSA (2 percent); after manual validation, the authors qualitatively analysed a stratified sample of 261 posts and 160 comment threads. Posts most often sought informational, therapeutic and relational help, and less often legal or technical help; threads most often offered information (72 of 160), then technical, relational and therapeutic advice (52 each), and least often institutional support.",
          "themes": [
            "Gender-based violence & misogyny",
            "Online sexual abuse & image-based abuse"
          ],
          "subtopics": [
            "Consent norms",
            "Deepfake pornography",
            "Gendered victimisation",
            "Intimate partner abuse",
            "Sextortion"
          ],
          "mdh_focus": [
            "H"
          ],
          "mdh_relevance": "direct",
          "relevance": 5,
          "stage": "Monitoring + Mitigation",
          "key_terms": [
            "Image-based sexual abuse",
            "Sextortion",
            "Help-seeking",
            "Victim-survivor support",
            "Nonconsensual synthetic explicit imagery"
          ],
          "models": [
            "text-bison",
            "text-unicorn"
          ],
          "method_qualifiers": [
            "Thematic analysis",
            "LLM-based filtering",
            "Zero-shot classification",
            "Multi-stage LLM prompt chaining"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "5.7M Reddit posts filtered to 113K; 261 posts and 160 comment threads coded.",
          "platform": [
            "Reddit"
          ],
          "region_country": [
            "United States"
          ],
          "targeted_group": [
            "Men",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Gender-based violence & misogyny": [
              "Gendered victimisation",
              "Intimate partner abuse"
            ],
            "Online sexual abuse & image-based abuse": [
              "Sextortion",
              "Deepfake pornography",
              "Consent norms"
            ]
          },
          "theme_qualifiers_canonical": {
            "Gender-based violence & misogyny": [
              "Gendered victimisation"
            ],
            "Online sexual abuse & image-based abuse": [
              "Consent norms",
              "Deepfake pornography"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "aberer",
      "name": "Karl Aberer",
      "url": "https://people.epfl.ch/karl.aberer",
      "unit": "LSIR",
      "faculty": "IC (emeritus, retired)",
      "mdh_focus": [
        "M",
        "D",
        "H"
      ],
      "dataTypes": [
        "Text",
        "Image"
      ],
      "techTypes": [
        "NLP",
        "Misinformation Detection",
        "Fact-Checking",
        "Bot Detection",
        "Source Credibility",
        "Knowledge Graphs",
        "Graph Neural Networks",
        "Social Media Analysis"
      ],
      "stage": "Prevention + Monitoring + Mitigation",
      "publications": [
        {
          "title": "Tactical Reframing of Online Disinformation Campaigns Against The Istanbul Convention",
          "wid": "tactical-reframing-of-online-disinformation-campaigns-a",
          "type": "publication",
          "year": 2021,
          "venue": "ICWSM 2021 Workshop on Data Mining for Online Misinformation and Disinformation (DWMV)",
          "link": "https://arxiv.org/abs/2105.13398",
          "authors": [
            "Tugrulcan Elmas (LSIR)",
            "Rebekah Overdorf (EPFL)",
            "Karl Aberer (LSIR)"
          ],
          "epfl_authors": [
            "Tugrulcan Elmas (LSIR)",
            "Rebekah Overdorf (EPFL)",
            "Karl Aberer (LSIR)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D",
            "H"
          ],
          "mdh_topics": [
            "Disinformation campaigns",
            "Narrative reframing",
            "Homophobia",
            "Gender-based violence",
            "Turkey",
            "Facebook",
            "Tactical reframing",
            "Astroturfing",
            "Cross-actor coordination"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "Empirical study tracing how an online disinformation campaign in Turkey shifted its message to build support for leaving the Istanbul Convention, the human-rights treaty on violence against women. Using public Facebook posts, it shows the campaign began in divorced men's groups complaining about the domestic implementing law, then was reframed to attack the convention by stressing its recognition of sexual orientation and non-traditional gender roles. Small men's-rights groups had their content amplified by larger political and religious pages and by a pro-government newspaper that shifted its own coverage the same way. It is presented as the first case study of narrative reframing inside a social-media disinformation campaign.",
          "why": "It documents a disinformation campaign that fused false framing with homophobic hate to roll back women's rights.",
          "data": "Text, Graph/Network (CrowdTangle public posts from ~2500 Turkish Facebook groups, pages, profiles)",
          "themes": [
            "Gender-based violence & misogyny",
            "Influence operations & coordinated manipulation",
            "Media framing & narrative analysis"
          ],
          "subtopics": [
            "Anti-gender campaigns",
            "Men's rights groups",
            "Narrative reframing",
            "Weaponised homophobia"
          ],
          "key_terms": [
            "Istanbul Convention",
            "Tactical reframing",
            "Homophobia",
            "Men's rights groups",
            "Turkey"
          ],
          "models": [],
          "method_qualifiers": [
            "Frame analysis",
            "Manual annotation",
            "Retrospective archive mining"
          ],
          "events_cases": [
            "Turkey's withdrawal from the Istanbul Convention"
          ],
          "built_at_epfl": [],
          "data_description": "Facebook posts from 2500 tracked Turkish groups, pages and profiles, 2014-2020.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Facebook"
          ],
          "region_country": [
            "Turkey"
          ],
          "targeted_group": [
            "LGBTQ+",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Gender-based violence & misogyny": [
              "Weaponised homophobia",
              "Anti-gender campaigns",
              "Men's rights groups"
            ],
            "Influence operations & coordinated manipulation": [
              "Grassroots campaign organizing",
              "Narrative reframing"
            ],
            "Media framing & narrative analysis": [
              "Narrative shift over time"
            ]
          },
          "theme_qualifiers_canonical": {
            "Gender-based violence & misogyny": [
              "Anti-gender campaigns",
              "Men's rights groups",
              "Weaponised homophobia"
            ],
            "Influence operations & coordinated manipulation": [
              "Grassroots campaign organizing",
              "Narrative reframing"
            ],
            "Media framing & narrative analysis": [
              "Issue framing",
              "Narrative shift over time"
            ]
          },
          "tentative": false
        },
        {
          "title": "SciLander: Mapping the Scientific News Landscape",
          "wid": "scilander-mapping-the-scientific-news-landscape",
          "type": "publication",
          "year": 2023,
          "venue": "ICWSM 2023 (Proc. International AAAI Conference on Web and Social Media, vol. 17, pp. 269-280)",
          "link": "https://ojs.aaai.org/index.php/ICWSM/article/view/22144",
          "authors": [
            "Maurício Gruppi (Rensselaer Polytechnic Institute)",
            "Panayiotis Smeros (EPFL)",
            "Sibel Adalı (Rensselaer Polytechnic Institute)",
            "Carlos Castillo (Universitat Pompeu Fabra)",
            "Karl Aberer (EPFL)"
          ],
          "epfl_authors": [
            "Panayiotis Smeros (EPFL)",
            "Karl Aberer (EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M"
          ],
          "mdh_topics": [
            "Scientific news veracity",
            "Source embeddings",
            "COVID-19 infodemic",
            "Citation stance",
            "News source reliability"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "Unsupervised method that learns embeddings of news sources from how they report on scientific topics, aiming to tell reliable from unreliable sources during the COVID-19 infodemic. It combines four source-level agreement signals (verbatim article copying, semantic shift in shared terms, use of scientific jargon, and the stance around cited references) to generate source pairs that train the embeddings. Evaluated on a large corpus of COVID-19 articles, it classifies source veracity at F1 = 87% and stays reliable online using only three months of activity. Clustering separates a politically unreliable group of sources from an alternative-health misinformation group.",
          "why": "It classifies the reliability of scientific news sources during the COVID-19 infodemic.",
          "data": "Text, Graph/Network (NELA-GT-2020 filtered to 991116 COVID-19 articles from 493 sources)",
          "themes": [
            "Media framing & narrative analysis",
            "Spread, amplification & networks",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Account coordination graphs",
            "Content-sharing networks",
            "Outlet credibility",
            "Rhetorical legitimation",
            "Shifting word meanings",
            "Source referencing practices"
          ],
          "key_terms": [
            "COVID-19",
            "Infodemic",
            "COVID-19 infodemic",
            "Conspiracy and pseudoscience sites",
            "Conspiracy theories"
          ],
          "models": [
            "BART",
            "BERT",
            "SciBERT",
            "Word2Vec"
          ],
          "method_qualifiers": [
            "Unsupervised representation learning",
            "Triplet loss",
            "Node embeddings",
            "Zero-shot classification"
          ],
          "events_cases": [
            "COVID-19 pandemic"
          ],
          "built_at_epfl": [
            {
              "name": "SciLander",
              "kind": "tool",
              "url": "https://github.com/mgruppi/SciLander",
              "evidence": "Paper, Reproducibility section: \"All the data, code, and models used for this paper are publicly available for research purposes in the following repository: https://github.com/mgruppi/SciLander.\" The repository README opens \"# SciLander: Mapping the Scientific News Landscape\" and describes its contents as \"Experiments: All the experiments we conducted for our submission to ICWSM 2023.\" and \"Model: SciLander models as well as baselines models that we used in our experiments.\""
            }
          ],
          "data_description": "1M COVID-19 news articles from 500 sources, 18 months.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "United States"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Media framing & narrative analysis": [
              "Shifting word meanings",
              "Rhetorical legitimation"
            ],
            "Spread, amplification & networks": [
              "Account coordination graphs",
              "Content-sharing networks"
            ],
            "Verification & content authenticity": [
              "Outlet credibility",
              "Source referencing practices"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media framing & narrative analysis": [
              "Rhetorical legitimation",
              "Shifting word meanings"
            ],
            "Spread, amplification & networks": [
              "Account coordination graphs"
            ],
            "Verification & content authenticity": [
              "Outlet credibility",
              "Source referencing practices"
            ]
          },
          "tentative": false
        },
        {
          "title": "Maximal fusion of facts on the web with credibility guarantee",
          "wid": "maximal-fusion-of-facts-on-the-web-with-credibility-gua",
          "type": "publication",
          "year": 2019,
          "venue": "Information Fusion (Elsevier), Vol. 48, pp. 55-66",
          "link": "https://doi.org/10.1016/j.inffus.2018.07.009",
          "authors": [
            "Thanh Tam Nguyen (Griffith University)",
            "Thanh Cong Phan",
            "Quoc Viet Hung Nguyen (Griffith University)",
            "Karl Aberer (EPFL)",
            "Bela Stantic (Griffith University)"
          ],
          "epfl_authors": [
            "Karl Aberer (EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M"
          ],
          "mdh_topics": [
            "Credibility",
            "Fact-checking",
            "Information fusion",
            "Web data",
            "Probabilistic graphical models",
            "Snopes"
          ],
          "stage": "Monitoring",
          "relevance": 4,
          "about": "Method for pulling factual information out of conflicting web sources while guaranteeing a user-chosen level of precision. It models the web as a network of sources, documents and claims, building a factor-graph that ties source features and document-level linguistic indicators to claim probabilities, and uses a credibility-reinforcement algorithm to return the largest set of claims that still meets the requested precision with low human-labelling effort. Evaluated on health, Snopes and Wiki datasets, it reliably hits the requested precision and returns about six times more claims than a comparable baseline at matched precision.",
          "why": "It is fact-extraction infrastructure evaluated on Snopes false-claim labels that supports fact-checking against misinformation.",
          "data": "Text, Graph/Network (Health 291K claims/11K sources, Snopes 4856 claims, Wiki 157 claims)",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Claim credibility inference",
            "Human fact-checking",
            "Precision guarantee"
          ],
          "key_terms": [
            "Truth discovery",
            "Claim credibility",
            "Information fusion",
            "Fact-checking",
            "Credibility extraction"
          ],
          "models": [],
          "method_qualifiers": [
            "Active learning",
            "Gibbs sampling",
            "Factor graph model",
            "Expectation-Maximization"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "3 datasets: Health (2.8M docs), Snopes (80K docs), Wikipedia (3K docs).",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Healthboards.com",
            "Snopes",
            "Wikipedia"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Claim credibility inference",
              "Human fact-checking",
              "Precision guarantee"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "Claim credibility inference",
              "Human fact-checking"
            ]
          },
          "tentative": false
        },
        {
          "title": "On Representation Learning for Scientific News Articles Using Heterogeneous Knowledge Graphs",
          "wid": "on-representation-learning-for-scientific-news-articles",
          "type": "publication",
          "year": 2021,
          "venue": "Companion Proceedings of The Web Conference 2021 (WWW 2021 Companion)",
          "link": "https://doi.org/10.1145/3442442.3451362",
          "authors": [
            "Angelika Romanou",
            "Panayiotis Smeros",
            "Karl Aberer"
          ],
          "epfl_authors": [
            "Angelika Romanou",
            "Panayiotis Smeros",
            "Karl Aberer"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M"
          ],
          "mdh_topics": [
            "Misinformation",
            "Fact-checking",
            "Scientific news",
            "Knowledge graphs",
            "Graph neural networks",
            "Credibility assessment"
          ],
          "stage": "Monitoring",
          "relevance": 4,
          "about": "Method that builds a heterogeneous knowledge graph over scientific news and tests whether graph neural networks can predict missing links in it, as a step toward credibility assessment. The graph links news articles, topics, cited papers, authors and institutions, and the study compares a structure-only baseline against content-aware models that add pretrained title embeddings as node features. On news-to-paper and news-to-topic link prediction the content-aware transformer model (HGT) consistently beats the structure-only baseline. It is presented as proof-of-concept evidence that combining article text with graph structure helps news credibility and fact-checking.",
          "why": "It is a representation-learning method for assessing the credibility of scientific news amid misinformation.",
          "data": "Text, Graph/Network (5569 entities, 9547 edges from NewsTeller)",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Claim credibility inference",
            "Source referencing practices"
          ],
          "key_terms": [
            "Scientific news credibility",
            "Heterogeneous knowledge graphs",
            "Graph neural networks",
            "Link prediction",
            "Misinformation"
          ],
          "models": [
            "HGT",
            "HetGNN",
            "R-GCN",
            "XLNet"
          ],
          "method_qualifiers": [
            "Graph embedding",
            "Link prediction"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "472 news articles, 1242 papers, 3464 authors, collected 2020.",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Source referencing practices",
              "Claim credibility inference"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "Claim credibility inference",
              "Source referencing practices"
            ]
          },
          "tentative": false
        },
        {
          "title": "Combating Online Scientific Misinformation (Doctoral Thesis)",
          "wid": "combating-online-scientific-misinformation",
          "type": "publication",
          "year": 2022,
          "venue": "EPFL Doctoral Thesis (Thèse n° 8256), IC Faculty, LSIR",
          "link": "https://infoscience.epfl.ch/record/294599",
          "authors": [
            "Panayiotis Smeros"
          ],
          "epfl_authors": [
            "Panayiotis Smeros"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "Scientific misinformation",
            "Fact-checking",
            "Claim detection",
            "News source credibility",
            "COVID-19",
            "Source embeddings"
          ],
          "stage": "Monitoring + Mitigation",
          "relevance": 5,
          "about": "Doctoral thesis surveying how online misinformation has evolved and presenting three systems for scientific misinformation at different levels of granularity. SciClops handles claim-level misinformation by extracting and clustering claims with related literature and ranking check-worthy ones; SciLens scores article-level quality from content, scientific context and social context; and SciLander learns source-level embeddings from signals such as article copying, semantic shift, jargon and citation stance. A final platform processes thousands of articles a day in real time. It reports these tools help fact-checkers verify scientific claims and judge article quality, with the COVID-19 infodemic as a focus.",
          "why": "It detects and contextualises scientific misinformation across the claim, article and source levels.",
          "data": "Text, Graph/Network",
          "themes": [
            "Media framing & narrative analysis",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Check-worthiness ranking",
            "Outlet credibility"
          ],
          "key_terms": [
            "Scientific misinformation",
            "Fact-checking",
            "News source credibility",
            "COVID-19 infodemic",
            "Claim verification"
          ],
          "models": [
            "BART",
            "BERT",
            "NewsBERT",
            "SciBERT",
            "SciNewsBERT",
            "Word2Vec",
            "XLNet"
          ],
          "method_qualifiers": [
            "Triplet margin loss",
            "Zero-shot classification",
            "Domain-specific fine-tuning",
            "Weak supervision"
          ],
          "events_cases": [
            "COVID-19 pandemic"
          ],
          "built_at_epfl": [
            {
              "name": "NewsBERT",
              "kind": "model",
              "url": "https://huggingface.co/psmeros/NewsBERT",
              "evidence": "Model card: 'Model Release for CIKM '21 paper: P. Smeros, C. Castillo, K. Aberer. SciClops: Detecting and Contextualizing Scientific Claims for Assisting Manual Fact-Checking.' SciClops is Chapter 4 of this thesis, and the thesis hub page links this exact model under 'NewsBERT and SciNewsBERT models, and validation set for scientific claim extraction'."
            },
            {
              "name": "NewsTeller",
              "kind": "platform",
              "url": "https://github.com/News-Teller",
              "evidence": "The Media Observatory Initiative page (PI Prof. Karl Aberer, LSIR, with Panayiotis Smeros on the team) lists under 'Project outputs': 'NewsTeller platform' and 'Source code (GitHub)', the latter linking to https://github.com/News-Teller. The org's own description is 'A research-driven platform to analyze news', located 'EPFL - Lausanne, CH'."
            },
            {
              "name": "SciClops",
              "kind": "tool",
              "url": "https://github.com/psmeros/SciClops",
              "evidence": "Repo README: 'Code Release for CIKM '21 paper: P. Smeros, C. Castillo, K. Aberer. SciClops: Detecting and Contextualizing Scientific Claims for Assisting Manual Fact-Checking.' The thesis hub page lists it under 'Code Release for SciClops, SciLens, and SciLander'."
            },
            {
              "name": "SciLander",
              "kind": "tool",
              "url": "https://github.com/psmeros/SciLander",
              "evidence": "Repo README: 'SciLander: Mapping the Scientific News Landscape ... Experiments: All the experiments we conducted for our submission to ICWSM 2023. ... Model: SciLander models as well as baselines models that we used in our experiments.' The thesis hub page links this repo under 'Code Release for SciClops, SciLens, and SciLander'."
            },
            {
              "name": "SciLens",
              "kind": "tool",
              "url": "https://github.com/psmeros/SciLens",
              "evidence": "Repo README: 'Code Release for WWW '19 paper: P. Smeros, C. Castillo, K. Aberer. SciLens: Evaluating the Quality of Scientific News Articles Using Social Media and Scientific Literature Indicators.'"
            },
            {
              "name": "SciNewsBERT",
              "kind": "model",
              "url": "https://huggingface.co/psmeros/SciNewsBERT",
              "evidence": "Model card: 'Model Release for CIKM '21 paper: P. Smeros, C. Castillo, K. Aberer. SciClops: Detecting and Contextualizing Scientific Claims for Assisting Manual Fact-Checking.'"
            },
            {
              "name": "NewsDiversifier",
              "kind": "tool",
              "url": "https://github.com/News-Teller/combat_echo",
              "evidence": "Repo README: 'We have developed a twitter bot that can be tagged directly on a tweet or on a reply to another tweet that contains a link to an article and will output 3 other articles that cover the same topic from different perspectives', and it sources its corpus 'by fetching all articles in a given timeframe from news-teller (http://newsteller.io/)'s ElasticSearch database'. Its reply template in src/main/twitter_core.py emits 'Hey @{tweet.user.screen_name}:' then 'url | medium bias | medium reliability' (or 'url | source | doi' when scientific=True) over three cutt.ly-shortened links - character-for-character the two @NewsDiversifier replies reproduced in Figure 7.5 of the thesis, including the #covid scientific variant."
            }
          ],
          "follow_up": [
            {
              "what": "The thesis's open-data hub: code, models, expert and non-expert annotations, the diffusion graph and the topic vocabularies, all released together",
              "kind": "dataset",
              "url": "https://psmeros.github.io/Combating_Online_Scientific_Misinformation.html",
              "evidence": "Page heading 'Combating Online Scientific Misinformation - Panayiotis Smeros, Carlos Castillo, Karl Aberer - PhD Thesis at EPFL', with a 'Code, Models, Datasets' section listing 'Code Release for SciClops, SciLens, and SciLander', 'NewsBERT and SciNewsBERT models, and validation set for scientific claim extraction', 'Anonymized evaluations of scientific claims by experts, non-experts, and commercial systems', 'Anonymized evaluations of scientific news articles by experts and non-experts', 'Anonymized social media postings, news articles, and scientific papers in graph format', 'Lists of Academic Repositories and Institution Domains', 'Vocabularies for Health and Nutrition, Science in News, and COVID-19'. Support credits: 'Open Science Fund' and 'Swiss Academy of Engineering Sciences'."
            },
            {
              "what": "Media Observatory Initiative, the IMI/OFCOM-funded LSIR project under which NewsTeller was built and operated",
              "kind": "project",
              "url": "https://www.media-initiative.ch/project/media-observatory-initiative/",
              "evidence": "'This project aims to develop a web platform to explore the Swiss and global news landscape ... The platform, called NewsTeller, captures, processes and references over 1.5 million news articles per month in 4 languages with improved context.' Team listed: 'Prof. Karl Aberer', 'Panayiotis Smeros', lab 'Distributed Information Systems Laboratory - LSIR'. 'This project has been co-sponsored by OFCOM.' 'This project started in October 2019 and has been completed'."
            },
            {
              "what": "Media Laboratory, the follow-on IMI-funded collaborative project awarded to the same team",
              "kind": "project",
              "url": "https://www.media-initiative.ch/project/media-laboratory/",
              "evidence": "Media Observatory Initiative page: 'The scientific team received funding from IMI for a new collaborative project called \"Media Laboratory\", which started in October 2020. In addition, a small extension to this project was developed between October and December 2022 as part of the \"CommPass\" project.'"
            },
            {
              "what": "SciLander was published after the thesis at ICWSM 2023 with two external collaborators (Gruppi, Adali) added",
              "kind": "successor-work",
              "url": "https://ojs.aaai.org/index.php/ICWSM/article/view/22144",
              "evidence": "Thesis hub page, Publications: '[ICWSM'23] M. Gruppi, P. Smeros, S. Adali, C. Castillo, K. Aberer. SciLander: Mapping the Scientific News Landscape.' The thesis (defended 1 July 2022) presents SciLander as its Chapter 6, and its acknowledgments name 'my external collaborators: Mauricio Gruppi and Sibel Adali, with whom we co-authored the paper SciLander'."
            }
          ],
          "data_description": "~1M news articles, social media reactions, scientific papers (CORD-19).",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "region_country": [
            "United States"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Outlet credibility",
              "Check-worthiness ranking"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "Check-worthiness ranking",
              "Outlet credibility"
            ]
          },
          "tentative": false
        },
        {
          "title": "Debunking Misinformation on the Web: Detection, Validation, and Visualisation (Doctoral Thesis)",
          "wid": "debunking-misinformation-on-the-web-detection-validatio",
          "type": "thesis",
          "year": 2019,
          "venue": "EPFL Doctoral Thesis (Thèse n° 9694), IC Faculty, LSIR",
          "link": "https://infoscience.epfl.ch/entities/publication/fa22b10d-4a6a-4556-a243-d24275d5c394",
          "authors": [
            "Thanh Tam Nguyen"
          ],
          "epfl_authors": [
            "Thanh Tam Nguyen"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "Rumour detection",
            "Social networks",
            "Human-in-the-loop validation",
            "Streaming data",
            "Anomaly detection"
          ],
          "stage": "Monitoring + Mitigation",
          "relevance": 5,
          "about": "Doctoral thesis proposing a three-part framework that pairs algorithmic models with human validators to address false content online. The detection part uses a graph-based progressive model to spot emerging misinformation stories in social-network data streams by exploiting the echo-chamber effect. The validation part designs guidance strategies that cut the human effort needed to confirm flags while raising confidence and lowering false alarms, including credibility-extraction factor-graph work. The visualisation part is a retention protocol that surfaces representative content for users overwhelmed by redundant information. It frames misinformation as an information-level cyber threat.",
          "why": "It is entirely about detecting, validating and visualising web misinformation and fake news at scale.",
          "data": "Text, Graph/Network",
          "themes": [
            "Spread, amplification & networks",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Account coordination graphs",
            "Claim credibility inference",
            "Echo chambers",
            "Human fact-checking",
            "Information cascades",
            "Validation effort minimisation"
          ],
          "key_terms": [
            "Rumour detection",
            "Fact-checking",
            "Crowdsourced validation",
            "Crowdsourcing",
            "Data stream summarisation"
          ],
          "models": [
            "CRF",
            "HDP"
          ],
          "method_qualifiers": [
            "Anomaly detection",
            "Graph scan statistics",
            "Causal inference",
            "Crowdsourced annotation"
          ],
          "events_cases": [
            "2016 U.S. presidential election",
            "Las Vegas shooting (2017)"
          ],
          "built_at_epfl": [
            {
              "name": "iCRF fact-checking validation framework",
              "kind": "framework"
            },
            {
              "name": "Minimal-regret data stream retaining algorithm",
              "kind": "tool"
            },
            {
              "name": "Twitter rumour detection dataset (4M tweets, 1022 rumours)",
              "kind": "dataset"
            }
          ],
          "follow_up": [
            {
              "what": "Detecting Rumours with Latency Guarantees using Massive Streaming Data (Technical Report, ACM, 2022; arXiv 2205.06580) - lifts the thesis's anomaly-based rumour detection from a static to a streaming setting, adding online pattern matching and coefficient-based load shedding under latency bounds, by the thesis author with two of the original co-authors",
              "kind": "successor-work",
              "url": "https://arxiv.org/abs/2205.06580",
              "evidence": "Conclusion: \"To this end, we took existing ideas on anomaly-based rumour detection, which identify local and global anomalies for propagation structures as captured by rumour patterns, as a starting point. Specifically, we lifted these ideas from a static setting to a streaming setting.\" The baseline it improves on is the thesis chapter itself: \"Static [51]: A static version of anomaly-based rumour detection, which is based on 45 rumour patterns.\" and \"Figure 3: Anomalies in the propagation structure, see [51].\" Reference [51] resolves to \"Nguyen Thanh Tam, Matthias Weidlich, Bolong Zheng, Hongzhi Yin, Nguyen Quoc Viet Hung, and Bela Stantic. 2019. From anomaly detection to rumour detection using data streams of social platforms. PVLDB 12, 9 (2019), 1016-1029\", which the thesis lists in 1.5 Selected Publications as one of the papers it is based on (Chapter 3, Detection)."
            }
          ],
          "data_description": "4M tweets, 4856 Snopes claims, healthcare claims; expert and crowd validation.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter"
          ],
          "region_country": [
            "United States"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Spread, amplification & networks": [
              "Echo chambers",
              "Information cascades",
              "Account coordination graphs"
            ],
            "Verification & content authenticity": [
              "Human fact-checking",
              "Claim credibility inference",
              "Validation effort minimisation"
            ]
          },
          "theme_qualifiers_canonical": {
            "Spread, amplification & networks": [
              "Account coordination graphs",
              "Echo chambers",
              "Information cascades"
            ],
            "Verification & content authenticity": [
              "Check-worthiness ranking",
              "Claim credibility inference",
              "Human fact-checking"
            ]
          },
          "tentative": false
        },
        {
          "title": "SciClops: Detecting and Contextualizing Scientific Claims for Assisting Manual Fact-Checking",
          "wid": "sciclops-detecting-and-contextualizing-scientific-claim",
          "type": "publication",
          "year": 2021,
          "venue": "CIKM 2021 (ACM International Conference on Information and Knowledge Management)",
          "link": "https://doi.org/10.1145/3459637.3482475",
          "authors": [
            "Panayiotis Smeros (LSIR, EPFL)",
            "Carlos Castillo (Universitat Pompeu Fabra)",
            "Karl Aberer (LSIR, EPFL)"
          ],
          "epfl_authors": [
            "Panayiotis Smeros (LSIR, EPFL)",
            "Karl Aberer (LSIR, EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M"
          ],
          "mdh_topics": [
            "Scientific misinformation",
            "Fact-checking",
            "Claim extraction",
            "Knowledge graph",
            "COVID-19",
            "NLP"
          ],
          "stage": "Monitoring + Mitigation",
          "relevance": 5,
          "about": "A three-step method that helps human fact-checkers verify scientific claims found in news and social media. It extracts candidate scientific claims with transformer models fine-tuned on scientific vocabulary, jointly clusters those claims with related scientific literature, then ranks check-worthy claims and builds an enhanced context of related verified claims, articles and papers. In a user study, members of the open public given this context scored closer to expert judgement than those with less context or two commercial fact-checking systems, reaching RMSE 1.02 against Google Fact Check's 2.79.",
          "why": "It is a fact-checking tool for scientific claims that outperforms commercial fact-checking systems in a user study.",
          "data": "Text, Graph/Network",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Evidence contextualisation",
            "Human fact-checking",
            "Source trustworthiness"
          ],
          "key_terms": [
            "Scientific misinformation",
            "Health misinformation",
            "Fact-checking",
            "Claim extraction",
            "Check-worthy claims"
          ],
          "models": [
            "BERT",
            "NewsBERT",
            "SciBERT",
            "SciNewsBERT"
          ],
          "method_qualifiers": [
            "Topic modeling",
            "Graph embedding",
            "Controlled experiment",
            "Crowdsourced annotation"
          ],
          "events_cases": [
            "COVID-19",
            "Marijuana and PTSD"
          ],
          "built_at_epfl": [
            {
              "name": "NewsBERT",
              "kind": "model",
              "url": "https://huggingface.co/psmeros/NewsBERT",
              "evidence": "Model card at huggingface.co/psmeros/NewsBERT: \"Model Release for CIKM '21 paper: P. Smeros, C. Castillo, K. Aberer. SciClops: Detecting and Contextualizing Scientific Claims for Assisting Manual Fact-Checking.\" The paper says: \"NewsBERT is a new model that we introduce, built on top of BERT and pretrained on a freely-available corpus of ~1M headlines published by the Australian Broadcasting Corporation\" and \"we make them publicly available for research purposes\"."
            },
            {
              "name": "SciClops",
              "kind": "tool",
              "url": "https://github.com/psmeros/SciClops",
              "evidence": "Repo README at github.com/psmeros/SciClops: \"Code Release for CIKM '21 paper: P. Smeros, C. Castillo, K. Aberer. SciClops: Detecting and Contextualizing Scientific Claims for Assisting Manual Fact-Checking.\""
            },
            {
              "name": "SciNewsBERT",
              "kind": "model",
              "url": "https://huggingface.co/psmeros/SciNewsBERT",
              "evidence": "Model card at huggingface.co/psmeros/SciNewsBERT: \"Model Release for CIKM '21 paper: P. Smeros, C. Castillo, K. Aberer. SciClops: Detecting and Contextualizing Scientific Claims for Assisting Manual Fact-Checking.\" The paper says: \"SciNewsBERT is also a new model that we introduce, pretrained like NewsBERT, albeit, it is built on top of SciBERT instead of BERT\"."
            }
          ],
          "data_description": "50K social media postings, 12K news articles, 24K scientific papers.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "News outlets",
            "Twitter"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Human fact-checking",
              "Evidence contextualisation",
              "Source trustworthiness"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "Evidence contextualisation",
              "Human fact-checking",
              "Outlet credibility"
            ]
          },
          "tentative": false
        },
        {
          "title": "The Role of Compromised Accounts in Social Media Manipulation (Doctoral Thesis)",
          "wid": "the-role-of-compromised-accounts-in-social-media-manipu",
          "type": "thesis",
          "year": 2022,
          "venue": "EPFL Doctoral Thesis (Thèse n° 8991), IC Faculty, LSIR",
          "link": "https://infoscience.epfl.ch/record/297318",
          "authors": [
            "Tuğrulcan Elmas"
          ],
          "epfl_authors": [
            "Tuğrulcan Elmas"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D",
            "H"
          ],
          "mdh_topics": [
            "Social media manipulation",
            "Compromised accounts",
            "Astroturfing",
            "Bot detection",
            "Twitter trends",
            "Account repurposing"
          ],
          "stage": "Monitoring + Mitigation",
          "relevance": 5,
          "about": "Doctoral thesis on how attackers weaponise compromised social-media accounts, structured around three contributions. It introduces ephemeral astroturfing, an attack that pushes a keyword or trend then deletes the activity so accounts can be reused. It shows that retweet bots bought on black markets are compromised real accounts rather than purpose-built ones, challenging prior bot-detection assumptions. It also builds a pipeline that finds accounts whose identity was changed to repurpose them while keeping their followers. The work detected over 19000 fake Twitter trends promoted by more than 108000 accounts.",
          "why": "It characterises how compromised accounts drive trend manipulation and disinformation campaigns.",
          "data": "Text, Graph/Network",
          "themes": [
            "Content moderation & enforcement",
            "Economics & incentives of MDH",
            "Influence operations & coordinated manipulation",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Account suspension analysis",
            "Fake trending topics",
            "Moderator-assist detection",
            "News-feed suppression",
            "Popularity mechanism manipulation"
          ],
          "key_terms": [
            "Ephemeral astroturfing",
            "Compromised accounts",
            "Retweet bots",
            "Misleading repurposing",
            "Astroturfing"
          ],
          "models": [
            "BERT"
          ],
          "method_qualifiers": [
            "Anomaly detection",
            "Supervised classification",
            "Honeypot infiltration"
          ],
          "events_cases": [
            "#SuriyelilerDefolsun campaign",
            "2016 U.S. election Russian interference"
          ],
          "built_at_epfl": [
            {
              "name": "Astrobot dataset",
              "kind": "dataset"
            },
            {
              "name": "Ephemeral astroturfing dataset",
              "kind": "dataset"
            },
            {
              "name": "EphemeralAstroturfing",
              "kind": "dataset"
            },
            {
              "name": "Misleading repurposing dataset",
              "kind": "dataset"
            },
            {
              "name": "Real-time fake trend detection Twitter bot",
              "kind": "tool"
            },
            {
              "name": "Retweet bot dataset",
              "kind": "dataset",
              "url": "https://github.com/tugrulz/RetweetBots",
              "evidence": "Thesis, Chapter 4 (Retweet Bots): \"The datasets are made available for reproducibility 1\", footnote \"1 https://github.com/tugrulz/RetweetBots\". Repository README: \"This repository contains the data described in Characterizing Retweet Bots: The Case of Black Market Accounts in dataset.csv\", split into the timeline dataset (\"These bots are not suspended (but probably not active), so you can readily collect them using Twitter's statuses/user_timeline endpoint.\") and the archive dataset (\"These are suspended, you need to collect them from Internet archive's Twitter dataset.\")."
            },
            {
              "name": "RetweetBots",
              "kind": "dataset"
            },
            {
              "name": "WayPop Machine",
              "kind": "tool",
              "url": "https://github.com/tugrulz/WayPop",
              "evidence": "Repository description: \"WayPop: A Wayback Machine to Investigate Popularity and Root Out Trolls\", forked from LSIR/Twitter-Time-Machine (LSIR being Karl Aberer's EPFL lab, the thesis's host lab). README: \"This repository contains the code to run the website for our application Twitter Time Machine, as well as the code to generate the data.\" and \"This application was created as part of our semester project at EPFL with LSIR. ... Developed by Thomas Ibanez & Alexandre Hutter.\" Thesis section 5.11 is titled \"WayPop Machine: A Wayback Machine to Investigate Repurposed Accounts\" and its Figure 5.8 caption describes exactly this stack: \"the processed data are stored in a NoSQL database, MongoDB. The web server built using the Django framework communicates with the data layer\". The two named developers are co-authors of the corresponding paper, Elmas, Ibanez, Hutter, Overdorf, Aberer, FOSINT-SI/ASONAM 2022."
            },
            {
              "name": "Ephemeral astroturfing attack dataset",
              "kind": "dataset",
              "url": "https://github.com/tugrulz/EphemeralAstroturfing",
              "evidence": "Thesis, Chapter 3 (Ephemeral Astroturfing), reproducibility statement: \"This research was conducted using the Internet Archive's Twitter Stream Grab and trends data, so all data is public. and the study is reproducible. In addition, the IDs of the tweets and users annotated in this study as well as the annotated attacks are made available 2\", footnote \"2 https://github.com/tugrulz/EphemeralAstroturfing\". Repository README: \"This repository contains the data, the annotations and the code for the paper 'Analyzing Activity and Suspension Patterns of Twitter Bots Attacking Turkish Twitter Trends by a Longitudinal Dataset' and 'Ephemeral Astroturfing Attacks: The Case of Fake Twitter Trends'.\" It ships fake_trends.csv, astrobot_annotations.csv and attack_annotations.csv."
            },
            {
              "name": "Repurposed accounts ground-truth dataset",
              "kind": "dataset",
              "url": "https://github.com/tugrulz/MisleadingRepurposing",
              "evidence": "Chapter 5 of the thesis is the paper 'Misleading Repurposing on Twitter' (Elmas, Overdorf, Aberer), whose published abstract ends: \"The data and the code is available at https://github.com/tugrulz/MisleadingRepurposing.\" The thesis states the artefact as contribution 3: \"establish a hand-labeled ground-truth dataset of repurposed accounts using datasets published by Twitter\"."
            }
          ],
          "follow_up": [
            {
              "what": "Chapter 5 of the thesis was published as 'Misleading Repurposing on Twitter' at ICWSM 2023 (Elmas, Overdorf, Aberer), a year after the thesis was defended, and that publication is what makes the ground-truth dataset and code public",
              "kind": "successor-work",
              "url": "https://ojs.aaai.org/index.php/ICWSM/article/view/22139",
              "evidence": "Article record: \"Misleading Repurposing on Twitter\", authors \"Tugrulcan Elmas, Rebekah Overdorf, Karl Aberer\", \"Vol. 17 (2023)\", published 2 June 2023. Abstract: \"We present the first in-depth and large-scale study of misleading repurposing ... We found over 100000 accounts that may have been repurposed. Of those, 28% were removed from the platform after 2 years, thereby confirming their inauthenticity. ... The data and the code is available at https://github.com/tugrulz/MisleadingRepurposing.\" The thesis version (2022) reports the 100000 figure but not the 28% two-year removal confirmation."
            },
            {
              "what": "Longitudinal astrobot dataset extending Chapter 3's ephemeral astroturfing detection from the thesis's 2019 annotation window to 212000+ bots and 29000 fake trends over 2015-2022, released into the thesis's own EphemeralAstroturfing repository (Elmas, WWW 2023 Companion)",
              "kind": "dataset",
              "url": "https://arxiv.org/abs/2304.07907",
              "evidence": "Abstract: \"Past work on such fake trends revealed a new astroturfing attack named ephemeral astroturfing that employs a very unique bot behavior in which bots post and delete generated tweets in a coordinated manner. As such, it is easy to mass-annotate such bots reliably, making them a convenient source of ground truth for bot research. In this paper, we detect and disclose over 212000 such bots targeting Turkish trends, which we name astrobots. ... We found that Twitter purged those bots en-masse 6 times since June 2018. ... The dataset is publicly available at https://github.com/tugrulz/EphemeralAstroturfing.\""
            }
          ],
          "data_description": "1% Twitter Stream Grab: 19485 fake trends, 108682 astrobots, 6199 retweet bots.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter"
          ],
          "region_country": [
            "Turkey",
            "United States"
          ],
          "targeted_group": [
            "LGBTQ+",
            "Syrian refugees",
            "migrants"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Moderator-assist detection",
              "News-feed suppression",
              "Account suspension analysis"
            ],
            "Influence operations & coordinated manipulation": [
              "Compromised account botnets",
              "Fake trending topics"
            ],
            "Spread, amplification & networks": [
              "Popularity mechanism manipulation"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Deplatforming",
              "Moderator-assist detection",
              "News-feed suppression"
            ],
            "Influence operations & coordinated manipulation": [
              "Compromised account botnets",
              "Ephemeral astroturfing"
            ],
            "Spread, amplification & networks": [
              "Account coordination graphs",
              "Popularity mechanism manipulation"
            ]
          },
          "tentative": false
        },
        {
          "title": "Harmful information used against humanitarian organisations (Engineering for Humanitarian Action / EHA)",
          "wid": "harmful-information-used-against-humanitarian-organisat",
          "type": "project",
          "year": "February 2021 - January 2024",
          "venue": "Engineering for Humanitarian Action initiative (ICRC + ETH Zurich + EPFL, launched December 2020)",
          "link": "https://eha.swiss/case-study/harmful-information-against-humanitarian-organisations/",
          "authors": [
            "Karl Aberer (LSIR, EPFL)",
            "Vincent Graf Narbel (ICRC)"
          ],
          "epfl_authors": [
            "Karl Aberer (LSIR, EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_topics": [
            "Humanitarian disinformation",
            "Anti-ICRC campaigns",
            "Ebola disinformation 2018",
            "Social media monitoring",
            "EHA partnership"
          ],
          "stage": "Prevention + Monitoring",
          "relevance": 5,
          "about": "Research project developing technical methods to monitor and counter disinformation aimed at humanitarian organisations and their field staff on social media. It was motivated by the 2018 Ebola response, when aid workers became targets of disinformation campaigns that led to violence. By studying how weaponised information is used against humanitarian operations and what the attack methods reveal, the team aims to help prevent future attacks and protect workers in the field. It is funded by the ICRC, the Stavros Niarchos Foundation, the Foundation for the ICRC, Rolex and the Fondation Lombard Odier.",
          "why": "It studies and counters disinformation campaigns targeting humanitarian organisations and their field staff.",
          "data": "Text, Social media (Twitter, Facebook) targeting humanitarian organisations and their workers",
          "lab": [
            "LSIR",
            "ICRC"
          ],
          "themes": [
            "Influence operations & coordinated manipulation",
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Attacks on aid organisations",
            "Disinformation campaigns",
            "Hateful rhetoric",
            "Identity-targeted hate",
            "Incitement to violence",
            "Information warfare"
          ],
          "key_terms": [
            "Humanitarian organisations",
            "Information warfare",
            "Aid worker safety",
            "Aid workers",
            "Disinformation campaigns"
          ],
          "models": [],
          "method_qualifiers": [],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "No dataset or scale stated (project description only).",
          "follow_up_checked": "2026-09-02",
          "targeted_group": [
            "Ethnic and religious groups",
            "Humanitarian aid workers"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Influence operations & coordinated manipulation": [
              "Information warfare",
              "Disinformation campaigns",
              "Attacks on aid organisations"
            ],
            "Toxicity & harassment": [
              "Identity-targeted hate",
              "Hateful rhetoric",
              "Incitement to violence"
            ]
          },
          "theme_qualifiers_canonical": {
            "Toxicity & harassment": [
              "Identity-targeted hate"
            ]
          },
          "tentative": true
        },
        {
          "title": "Can Celebrities Burst Your Bubble?",
          "wid": "can-celebrities-burst-your-bubble",
          "type": "publication",
          "year": 2020,
          "venue": "arXiv 2003.06857; WWW 2020 Companion (Innovative Ideas in Data Science workshop)",
          "link": "https://arxiv.org/abs/2003.06857",
          "authors": [
            "Tuğrulcan Elmas",
            "Kristina Hardi",
            "Rebekah Overdorf",
            "Karl Aberer"
          ],
          "epfl_authors": [
            "Tuğrulcan Elmas",
            "Kristina Hardi",
            "Rebekah Overdorf",
            "Karl Aberer"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Filter bubbles",
            "Polarization",
            "Random Walk Controversy",
            "Celebrities-as-bridges",
            "Echo chambers",
            "Counter-narrative"
          ],
          "stage": "Mitigation",
          "relevance": 2,
          "about": "A short paper proposing celebrities as a way to reduce online political polarization. The idea is to recommend polarizing or contrarian topics to celebrities so that, when they weigh in, their followers on both sides of a debate are exposed to opposing viewpoints. It frames the choice of which accounts to involve as finding ones that are both popular and politically neutral, and tests this on a Turkish election case study. Adding popular and neutral celebrities works far better than adding only the most popular accounts, and the effect holds even when a celebrity loses up to 80 percent of their followers.",
          "why": "It is a polarization-reduction intervention treating polarization and filter bubbles as threats to democratic discourse.",
          "data": "Text + Graph (Twitter follower data); case study on 2019 Istanbul Election Rerun + 81 Turkish celebrities; #Russia_March topic for empirical polarization model",
          "themes": [
            "Persuasion & cognitive effects",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Echo chambers",
            "Political polarisation",
            "Spread interventions"
          ],
          "key_terms": [
            "Filter bubbles",
            "Celebrity influence",
            "Polarization",
            "Echo chambers",
            "2019 Istanbul election rerun"
          ],
          "models": [],
          "method_qualifiers": [
            "Random walk simulation",
            "Manual annotation"
          ],
          "events_cases": [
            "#Russia_March Twitter debate",
            "2019 Istanbul Election Rerun"
          ],
          "built_at_epfl": [],
          "data_description": "Twitter follower graphs and 679 replies to 81 celebrity tweets, Istanbul 2019.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "region_country": [
            "Russia",
            "Turkey"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Spread, amplification & networks": [
              "Echo chambers",
              "Spread interventions",
              "Political polarisation"
            ]
          },
          "theme_qualifiers_canonical": {
            "Spread, amplification & networks": [
              "Echo chambers",
              "Political polarisation",
              "Spread interventions"
            ]
          },
          "tentative": false
        },
        {
          "title": "WayPop Machine: A Wayback Machine to Investigate Popularity and Root Out Trolls",
          "wid": "waypop-machine-a-wayback-machine-to-investigate-popular",
          "type": "publication",
          "year": 2022,
          "venue": "ASONAM 2022 (IEEE/ACM Int. Conf. on Advances in Social Networks Analysis and Mining)",
          "link": "https://www.computer.org/csdl/proceedings-article/asonam/2022/10068665/1LKx6Psx6ve",
          "authors": [
            "Tugrulcan Elmas",
            "Thomas Romain Ibanez",
            "Alexandre Hutter",
            "Rebekah Overdorf",
            "Karl Aberer"
          ],
          "epfl_authors": [
            "Tugrulcan Elmas",
            "Thomas Romain Ibanez",
            "Alexandre Hutter",
            "Rebekah Overdorf",
            "Karl Aberer"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Popularity manipulation",
            "Fake amplification",
            "Trolls",
            "Bot detection",
            "Wayback Machine",
            "Social media manipulation"
          ],
          "stage": "Monitoring",
          "relevance": 4,
          "about": "A method and open-source tool that works out why a social-media account became popular. The motivation is that malicious users such as trolls must manufacture their popularity on the platform itself, often illicitly through fake amplification, unlike celebrities whose fame comes from offline activity. By reconstructing an account's follower and popularity history through the Wayback Machine, the tool helps tell apart accounts that grew honestly from those whose influence was bought or faked.",
          "why": "It detects inauthentically amplified accounts and trolls, supporting disinformation investigation.",
          "data": "Text, social-media follower/popularity histories via the Wayback Machine",
          "themes": [
            "Influence operations & coordinated manipulation",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Account repurposing",
            "Influential-user tracking",
            "Information cascades",
            "Popularity mechanism manipulation"
          ],
          "key_terms": [
            "OSINT",
            "Misleading repurposing",
            "Follower growth",
            "Trolls",
            "Twitter"
          ],
          "models": [],
          "method_qualifiers": [
            "Anomaly detection",
            "Retrospective archive mining"
          ],
          "events_cases": [
            "TDP Twitter account repurposing",
            "TDP account compromise"
          ],
          "built_at_epfl": [
            {
              "name": "WayPop",
              "kind": "tool",
              "url": "https://github.com/tugrulz/WayPop",
              "evidence": "Paper: \"WayPop is publicly available on GitHub at https://github.com/tugrulz/WayPop.\" The repository description reads \"A Wayback Machine to Investigate Popularity and Root Out Trolls\", and its README states \"This application was created as part of our semester project at EPFL with LSIR\" and credits \"Thomas Ibanez & Alexandre Hutter\", two of the paper's co-authors."
            }
          ],
          "data_description": "4.67 TB Twitter Stream Grab, 1% sample, popular users >5k followers.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "region_country": [
            "Turkey",
            "United States"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Influence operations & coordinated manipulation": [
              "Account repurposing"
            ],
            "Spread, amplification & networks": [
              "Popularity mechanism manipulation",
              "Influential-user tracking",
              "Information cascades"
            ]
          },
          "theme_qualifiers_canonical": {
            "Influence operations & coordinated manipulation": [
              "Account repurposing"
            ],
            "Spread, amplification & networks": [
              "Influential-user tracking",
              "Information cascades",
              "Popularity mechanism manipulation"
            ]
          },
          "tentative": false
        },
        {
          "title": "Characterizing Retweet Bots: The Case of Black Market Accounts",
          "wid": "characterizing-retweet-bots-the-case-of-black-market-ac",
          "type": "publication",
          "year": 2022,
          "venue": "ICWSM 2022 (Sixteenth International AAAI Conference on Web and Social Media); arXiv 2112.02366",
          "link": "https://arxiv.org/abs/2112.02366",
          "authors": [
            "Tuğrulcan Elmas",
            "Rebekah Overdorf",
            "Karl Aberer"
          ],
          "epfl_authors": [
            "Tuğrulcan Elmas",
            "Rebekah Overdorf",
            "Karl Aberer"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Retweet bots",
            "Black-market accounts",
            "Compromised accounts",
            "Bot detection challenges",
            "Twitter",
            "Inauthentic amplification"
          ],
          "stage": "Monitoring",
          "relevance": 4,
          "about": "The first study to focus specifically on retweet bots, the accounts paid to amplify tweets. Instead of relying on error-prone human labelling, the authors obtained reliable examples by directly purchasing retweet services from black-market vendors, then characterised how the bots behave over their lifecycle against human-annotated controls. A central finding is that many retweet bots are not freshly mass-created accounts but compromised genuine accounts driven aggressively by an attacker, so several assumptions in earlier bot-detection work do not hold up.",
          "why": "It is a bot-detection study on inauthentic amplification that overturns prior assumptions in disinformation work.",
          "data": "Text + Account metadata (Twitter); 862 non-suspended retweet bots + 5332 suspended bots purchased from black market (extending Golbeck 2019); control groups of 27622 human-annotated accounts; 1.2M retweets + 126K tweets in timeline dataset; 302K retweets + 30K tweets in Internet Archive Stream Grab dataset",
          "themes": [
            "Economics & incentives of MDH",
            "Influence operations & coordinated manipulation",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Black-market accounts",
            "Bot amplification",
            "Purchased engagement",
            "Retweet timing patterns"
          ],
          "key_terms": [
            "Retweet bots",
            "Account compromise",
            "Black market accounts",
            "Astroturfing",
            "Bot detection"
          ],
          "models": [],
          "method_qualifiers": [
            "Wordshift analysis",
            "Welch's t-test",
            "Retrospective archive mining"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "RetweetBots dataset",
              "kind": "dataset"
            },
            {
              "name": "RetweetBots",
              "kind": "dataset",
              "url": "https://github.com/tugrulz/RetweetBots",
              "evidence": "From the paper's own full text: 'The datasets are made available for reproducibility 1 .' with footnote '1 https://github.com/tugrulz/RetweetBots'. The repository exists and its readme cites 'Characterizing Retweet Bots: The Case of Black Market Accounts' (Elmas, Overdorf, Aberer, ICWSM 2022); it holds a CSV of account ids split into the timeline subset (active or unsuspended accounts, collectable via Twitter statuses/user_timeline) and the archive subset (suspended accounts, via the Internet Archive Twitter Stream Grab), with full data available on request for research use."
            }
          ],
          "data_description": "6199 retweet-bot accounts and 27622 human accounts, about 1.5M retweets.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Economics & incentives of MDH": [
              "Black-market accounts",
              "Purchased engagement"
            ],
            "Influence operations & coordinated manipulation": [
              "Bot amplification",
              "Compromised account botnets",
              "Black-market retweets"
            ],
            "Spread, amplification & networks": [
              "Retweet timing patterns"
            ]
          },
          "theme_qualifiers_canonical": {
            "Economics & incentives of MDH": [
              "Black-market accounts",
              "Purchased engagement"
            ],
            "Influence operations & coordinated manipulation": [
              "Bot amplification",
              "Compromised account botnets"
            ],
            "Spread, amplification & networks": [
              "Account coordination graphs",
              "Popularity mechanism manipulation"
            ]
          },
          "tentative": false
        },
        {
          "title": "Ephemeral Astroturfing Attacks: The Case of Fake Twitter Trends",
          "wid": "ephemeral-astroturfing-attacks-the-case-of-fake-twitter",
          "type": "publication",
          "year": 2021,
          "venue": "IEEE European Symposium on Security and Privacy (EuroS&P) 2021",
          "link": "https://doi.org/10.1109/EuroSP51992.2021.00041",
          "authors": [
            "Tuğrulcan Elmas",
            "Rebekah Overdorf",
            "Ahmed Furkan Özkalay",
            "Karl Aberer"
          ],
          "epfl_authors": [
            "Tuğrulcan Elmas",
            "Rebekah Overdorf",
            "Ahmed Furkan Özkalay",
            "Karl Aberer"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D",
            "H"
          ],
          "mdh_topics": [
            "Astroturfing",
            "Fake Twitter trends",
            "Compromised accounts",
            "Deletion-after-promotion",
            "Real-time detection"
          ],
          "stage": "Monitoring + Mitigation",
          "relevance": 5,
          "about": "A paper that discovers and characterises a manipulation technique called ephemeral astroturfing, in which coordinated inauthentic accounts push a keyword onto Twitter's trending list and then delete the activity so the compromised accounts can be reused without being flagged. Working from an archived sample of the Twitter stream, the team finds more than 19000 unique fake trends promoted by over 108000 accounts, with ephemerally astroturfed trends making up at least 20 percent of the top 10 global trends during the observation period. The authors released a real-time detector.",
          "why": "It uncovers a novel disinformation attack affecting a fifth of top global trends and ships a live detector.",
          "data": "Twitter Stream Grab (Internet Archive 1% sample); 19000+ unique fake trends; 108000+ accounts; real-time detection bot",
          "themes": [
            "Economics & incentives of MDH",
            "Influence operations & coordinated manipulation",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Black-market accounts",
            "Ephemeral astroturfing",
            "Fake follower schemes",
            "Popularity mechanism manipulation"
          ],
          "key_terms": [
            "Ephemeral astroturfing",
            "Compromised accounts",
            "Turkey",
            "Twitter trends",
            "Coordinated inauthentic behavior"
          ],
          "models": [],
          "method_qualifiers": [
            "Honeypot infiltration",
            "Decision tree",
            "Black-box auditing",
            "Community detection"
          ],
          "events_cases": [
            "#SuriyelilerDefolsun anti-refugee campaign",
            "SuriyelilerDefolsun hashtag",
            "Turkish Local Elections 2019"
          ],
          "built_at_epfl": [
            {
              "name": "Annotated ephemeral astroturfing attacks",
              "kind": "dataset"
            },
            {
              "name": "EphemeralAstroturfing",
              "kind": "dataset",
              "url": "https://github.com/tugrulz/EphemeralAstroturfing",
              "evidence": "Paper, 'Ethics and Reproducibility': \"In addition, the IDs of the tweets and users annotated in this study as well as the annotated attacks are made available 3\", footnote 3 being \"https://github.com/tugrulz/EphemeralAstroturfing\". Repository README: \"This repository contains the data, the annotations and the code for the paper ... Ephemeral Astroturfing Attacks: The Case of Fake Twitter Trends\", citing elmas2021ephemeral (Elmas, Overdorf, Özkalay, Aberer, IEEE EuroS&P 2021, pp. 403-422). It ships the annotated fake trends, tweet/user IDs with deletion timestamps, Botometer scores, and lexicon_classifier.py, \"Rule based classifier to detect lexicon tweets\"."
            }
          ],
          "follow_up": [
            {
              "what": "Longitudinal astrobot dataset extending the attack's detection method to 212000+ bots (Elmas, 'Analyzing Activity and Suspension Patterns of Twitter Bots Attacking Turkish Twitter Trends by a Longitudinal Dataset', WWW 2023 Companion), released into this paper's own repository",
              "kind": "successor-work",
              "url": "https://arxiv.org/abs/2304.07907",
              "evidence": "Abstract: \"Past work on such fake trends revealed a new astroturfing attack named ephemeral astroturfing that employs a very unique bot behavior in which bots post and delete generated tweets in a coordinated manner.\" It uses that behaviour as ground truth to detect and disclose over 212000 bots ('astrobots') targeting Turkish trends from June 2018 onward, and states: \"The dataset is publicly available at https://github.com/tugrulz/EphemeralAstroturfing.\""
            }
          ],
          "data_description": "32895 attacked trends, 108000 accounts, Twitter data 2015-2019.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "region_country": [
            "Turkey"
          ],
          "targeted_group": [
            "LGBTQ+",
            "Migrants",
            "Political candidates",
            "Refugees",
            "Syrians"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Economics & incentives of MDH": [
              "Astroturfing as a service",
              "Black-market accounts",
              "Fake follower schemes"
            ],
            "Influence operations & coordinated manipulation": [
              "Ephemeral astroturfing",
              "Compromised account botnets"
            ],
            "Spread, amplification & networks": [
              "Popularity mechanism manipulation"
            ]
          },
          "theme_qualifiers_canonical": {
            "Economics & incentives of MDH": [
              "Astroturfing as a service",
              "Black-market accounts",
              "Fake follower schemes"
            ],
            "Influence operations & coordinated manipulation": [
              "Compromised account botnets",
              "Ephemeral astroturfing"
            ],
            "Spread, amplification & networks": [
              "Account coordination graphs",
              "Information cascades",
              "Popularity mechanism manipulation"
            ]
          },
          "tentative": false
        },
        {
          "title": "A Dataset of State-Censored Tweets",
          "wid": "a-dataset-of-state-censored-tweets",
          "type": "publication",
          "year": 2021,
          "venue": "ICWSM 2021 (International AAAI Conference on Web and Social Media)",
          "link": "https://ojs.aaai.org/index.php/ICWSM/article/view/18124",
          "authors": [
            "Tuğrulcan Elmas",
            "Rebekah Overdorf",
            "Karl Aberer"
          ],
          "epfl_authors": [
            "Tuğrulcan Elmas",
            "Rebekah Overdorf",
            "Karl Aberer"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D",
            "H"
          ],
          "mdh_topics": [
            "State censorship",
            "Twitter",
            "Censored content dataset",
            "Information control"
          ],
          "stage": "Monitoring",
          "relevance": 3,
          "about": "A paper releasing a public dataset of tweets that governments asked Twitter to withhold. It exploits the fact that Twitter withholds content regionally rather than deleting it everywhere, so censored material can still be collected from outside the affected region using archived Twitter data. The authors point to uses such as studying government censorship, detecting hate speech, and measuring the effect of censorship on users.",
          "why": "It enables research on state-level information control and censorship, with hate-speech detection noted as a use case.",
          "data": "Twitter; dataset of state-censored tweets",
          "themes": [
            "Content moderation & enforcement",
            "Platform governance & regulation",
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Censorship policies",
            "Country-withheld content",
            "Legal removal requests",
            "State-ordered takedowns"
          ],
          "key_terms": [
            "Government censorship",
            "Country-withheld content",
            "Hate speech detection",
            "Twitter",
            "Dataset"
          ],
          "models": [],
          "method_qualifiers": [
            "Descriptive statistics",
            "Social network extension",
            "Retrospective archive mining"
          ],
          "events_cases": [
            "Kashmir dispute",
            "Operation Olive Branch"
          ],
          "built_at_epfl": [
            {
              "name": "A Dataset of State-Censored Tweets",
              "kind": "dataset"
            },
            {
              "name": "CensoredTweets",
              "kind": "dataset"
            },
            {
              "name": "Dataset of State-Censored Tweets",
              "kind": "dataset",
              "url": "https://doi.org/10.5281/zenodo.4439509",
              "evidence": "Paper abstract: 'The dataset is publicly available at https://doi.org/10.5281/zenodo.4439509' and Section 1: 'We made the dataset available at Zenodo: https://doi.org/10.5281/zenodo.4439509. The dataset only consists of tweet ids and user ids in order to comply with Twitter Terms of Service.' The Zenodo record is titled 'A Dataset of State-Censored Tweets' by Tugrulcan Elmas, Rebekah Overdorf and Karl Aberer (EPFL), states it 'is the dataset associated with the paper of the same name', references arxiv.org/abs/2101.05919, and was published 14 January 2021 under CC-BY-4.0."
            }
          ],
          "follow_up": [
            {
              "what": "CensoredTweets code repository, the reproduction pipeline released alongside the dataset",
              "kind": "repository",
              "url": "https://github.com/tugrulz/CensoredTweets",
              "evidence": "Paper, Section 1: 'For the documentation and the code to reproduce the pipeline please refer to https://github.com/tugrulz/CensoredTweets.' The repository README states it provides 'documentation and the code for reproduction of the paper \"A Dataset of State-Censored Tweets\"', cites 'Elmas, Tugrulcan, Overdorf, Rebekah and Aberer, Karl. \"A Dataset of State-Censored Tweets.\" arXiv preprint arXiv:2101.05919 (2021)', and links the dataset at https://zenodo.org/record/4439509."
            },
            {
              "what": "State & Geopolitical Censorship on Twitter (X): Detection & Impact Analysis of Withheld Content, CIKM 2025, by Cetinkaya and Elmas - the first quantitative impact analysis built on this dataset",
              "kind": "successor-work",
              "url": "https://arxiv.org/abs/2508.13375",
              "evidence": "From the paper's Data section: 'Censorships until 2021: The first dataset is collected using the Internet Archive's Twitter Stream Grab, which contains 1% of all tweets between September 2011 and June 2020, from which all tweets with a non-empty \"withheld in countries\" field are extracted, yielding 583437 censored tweets from 155715 unique users [8]. Fully censored accounts are identified via the Twitter User Lookup API and an inference heuristic, resulting in 4301 entirely withheld users.' Reference [8] is 'Tugrulcan Elmas, Rebekah Overdorf, and Karl Aberer. 2021. A dataset of state-censored tweets. In Proceedings of the International AAAI Conference on Web and Social Media, Vol. 15. 1009-1015.'"
            }
          ],
          "data_description": "583k censored tweets, 155k users, 22m supplemental tweets, 2012-2020.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "region_country": [
            "France",
            "Germany",
            "India",
            "Russia",
            "Turkey"
          ],
          "targeted_group": [
            "Ethnic minorities",
            "Muslims",
            "Political dissidents",
            "Religious minorities"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "State-ordered takedowns",
              "Legal removal requests",
              "Country-withheld content"
            ],
            "Platform governance & regulation": [
              "Censorship policies",
              "State-ordered takedowns",
              "Country-withheld content"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "State-ordered takedowns"
            ],
            "Platform governance & regulation": [
              "Censorship policies"
            ]
          },
          "tentative": false
        },
        {
          "title": "Misleading Repurposing on Twitter",
          "wid": "misleading-repurposing-on-twitter",
          "type": "publication",
          "year": 2023,
          "venue": "ICWSM 2023 (Seventeenth International AAAI Conference on Web and Social Media)",
          "link": "https://ojs.aaai.org/index.php/ICWSM/article/view/22139",
          "authors": [
            "Tuğrulcan Elmas (EPFL)",
            "Rebekah Overdorf (UNIL)",
            "Karl Aberer (EPFL)"
          ],
          "epfl_authors": [
            "Tuğrulcan Elmas (EPFL)",
            "Rebekah Overdorf (UNIL)",
            "Karl Aberer (EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Misleading repurposing",
            "Account identity change",
            "High-follower targeting",
            "Twitter manipulation",
            "Public detection tool"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "The first large-scale study of misleading repurposing, where someone changes an account's identity by altering profile attributes such as handle, name and bio so the account can be reused for a new purpose while keeping its existing followers. The team detects these accounts with a supervised machine-learning pipeline applied to historical Twitter data and releases a public tool to flag popular accounts that were later repurposed. It identifies more than 100000 potentially repurposed accounts, and finds adversaries tend to target accounts with large follower counts and repurpose them after a period of inactivity and tweet deletion.",
          "why": "It names and provides a detector for a concrete account-manipulation tactic used to mislead audiences.",
          "data": "Twitter Stream Grab (Internet Archive 1% sample); supervised-learning pipeline; >100000 detected repurposed accounts",
          "themes": [
            "Content moderation & enforcement",
            "Economics & incentives of MDH",
            "Fraud, impersonation & forgery",
            "Influence operations & coordinated manipulation"
          ],
          "subtopics": [
            "Account repurposing",
            "Coordinated propaganda networks"
          ],
          "key_terms": [
            "Misleading repurposing",
            "Fake account trafficking",
            "Follow-back schemes",
            "Twitter",
            "Account identity change"
          ],
          "models": [
            "BERT",
            "bert-base-multilingual-uncased"
          ],
          "method_qualifiers": [
            "Style change detection",
            "Active learning",
            "Supervised classification"
          ],
          "events_cases": [
            "2016 U.S. elections",
            "Brexit"
          ],
          "built_at_epfl": [
            {
              "name": "Misleading Repurposing dataset and classifier",
              "kind": "dataset"
            },
            {
              "name": "MisleadingRepurposing",
              "kind": "dataset",
              "url": "https://github.com/tugrulz/MisleadingRepurposing",
              "evidence": "Last line of the paper's own abstract, verbatim: \"The data and the code is available at https://github.com/tugrulz/MisleadingRepurposing.\" The same line is repeated on the publisher's record at https://ojs.aaai.org/index.php/ICWSM/article/view/22139 (DOI 10.1609/icwsm.v17i1.22139, pages 209-220), which lists the authors as Tugrulcan Elmas and Karl Aberer (EPFL) and Rebekah Overdorf (University of Lausanne)."
            },
            {
              "name": "Repurposed accounts ground-truth dataset",
              "kind": "dataset"
            }
          ],
          "data_description": "Historical Twitter profile snapshots and tweets from 446M user IDs, 2011-2020.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Influence operations & coordinated manipulation": [
              "Account repurposing",
              "Coordinated propaganda networks"
            ]
          },
          "theme_qualifiers_canonical": {
            "Influence operations & coordinated manipulation": [
              "Account repurposing",
              "Coordinated propaganda networks"
            ]
          },
          "tentative": false
        },
        {
          "title": "Disinformation from the Inside, Combining Machine Learning and Journalism to Investigate Sockpuppet Campaigns",
          "wid": "disinformation-from-the-inside-combining-machine-learni",
          "type": "publication",
          "year": 2020,
          "link": "https://doi.org/10.1145/3366424.3385777",
          "authors": [
            "Christopher Schwartz (KU Leuven)",
            "Rebekah Overdorf (EPFL LSIR)"
          ],
          "epfl_authors": [
            "Rebekah Overdorf (EPFL LSIR)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Sockpuppet campaigns",
            "Disinformation typology"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "Study combining machine learning with investigative journalism to examine a sockpuppet disinformation campaign in Kyrgyzstan, with tweet collection oriented to the October 2017 presidential election. Sockpuppets are human-controlled fake accounts, and the paper argues that as bot detection improves, adversaries shift toward them, especially infiltrators that integrate into a target community to persuade it from within. It sets out the infiltrator as a subset of sockpuppets, set against bots, whose chief effect the authors describe as driving traffic or drowning out opposition while infiltrators assimilate genuine audiences from within. The Kyrgyz case draws on a whistleblower who described the sockpuppets as \"high quality\", which the authors gloss as few in number but resource-intensive; separately the authors estimate that the average size of a sockpuppet campaign in Kyrgyzstan ranges between 10 and 100 fake accounts, and the typology grounds later EPFL work on astroturfing and Telegram propaganda accounts.",
          "why": "It names the infiltrator as an adversarial escalation of meatpuppets and Sybils, and documents a real sockpuppet disinformation campaign.",
          "data": "Network",
          "themes": [
            "Influence operations & coordinated manipulation",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Sockpuppet accounts"
          ],
          "key_terms": [
            "Kyrgyzstan",
            "Investigative journalism",
            "Sockpuppet campaigns",
            "Infiltrator accounts",
            "Bot-detection evasion"
          ],
          "models": [],
          "method_qualifiers": [
            "Journalist-in-the-loop supervision",
            "Supervised classification",
            "Manual annotation",
            "Adversarial evasion"
          ],
          "events_cases": [
            "2017 Kyrgyz presidential election",
            "Kyrgyzstan 2017 presidential election",
            "Kyrgyzstan sockpuppet disinformation campaigns",
            "RE:AKIA anti-corruption protests"
          ],
          "built_at_epfl": [],
          "data_description": "Facebook and Twitter accounts from Kyrgyzstan, manually annotated; one whistleblower source.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Facebook",
            "Twitter/X"
          ],
          "region_country": [
            "Kyrgyzstan"
          ],
          "targeted_group": [
            "Ethnic minorities",
            "Political opposition"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Influence operations & coordinated manipulation": [
              "Sockpuppet accounts"
            ]
          },
          "theme_qualifiers_canonical": {
            "Influence operations & coordinated manipulation": [
              "Sockpuppet accounts"
            ]
          },
          "tentative": true
        },
        {
          "title": "SciLens News Platform: A System for Real-Time Evaluation of News Articles",
          "wid": "scilens-news-platform-a-system-for-real-time-evaluation",
          "type": "publication",
          "year": 2020,
          "venue": "Proceedings of the VLDB Endowment 13(12), 2020",
          "link": "https://arxiv.org/abs/2008.12039",
          "authors": [
            "Angelika Romanou",
            "Panayiotis Smeros",
            "Carlos Castillo",
            "Karl Aberer"
          ],
          "epfl_authors": [
            "Angelika Romanou",
            "Panayiotis Smeros",
            "Karl Aberer"
          ],
          "about": "Demonstration of the SciLens News Platform, which collects contextual information about news articles in real time and provides indicators of their validity and trustworthiness. The indicators draw on social media discussions of each article (reach and stance) and on its content and referenced sources, and are combined with expert reviews on seven criteria such as factual accuracy. In a COVID-19 use case over 45 mainstream outlets, low-quality outlets dedicated a larger share of their articles to the topic from the end of the first month and tended to acquire more social media reach, while high-quality outlets based their findings more on scientific references.",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Human fact-checking",
            "Outlet credibility",
            "Source referencing practices"
          ],
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "Monitoring + Mitigation",
          "key_terms": [
            "News quality evaluation",
            "COVID-19",
            "Computational journalism",
            "Expert reviews",
            "Clickbait"
          ],
          "models": [],
          "method_qualifiers": [
            "Topic modeling",
            "Stance analysis",
            "Manual annotation",
            "Stream processing"
          ],
          "events_cases": [
            "COVID-19 pandemic"
          ],
          "built_at_epfl": [
            {
              "name": "SciLens News Platform",
              "kind": "platform"
            }
          ],
          "data_description": "45 rated science outlets' COVID-19 articles, 60 days, 2020-01-15 to 2020-03-15.",
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Source referencing practices",
              "Outlet credibility",
              "Human fact-checking"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "Human fact-checking",
              "Outlet credibility",
              "Source referencing practices"
            ]
          },
          "tentative": false
        },
        {
          "title": "SciLens: Evaluating the Quality of Scientific News Articles Using Social Media and Scientific Literature Indicators",
          "wid": "scilens-evaluating-the-quality-of-scientific-news-artic",
          "type": "publication",
          "year": 2019,
          "venue": "The Web Conference (WWW) 2019",
          "link": "https://arxiv.org/abs/1903.05538",
          "authors": [
            "Panayiotis Smeros",
            "Carlos Castillo",
            "Karl Aberer"
          ],
          "epfl_authors": [
            "Panayiotis Smeros",
            "Karl Aberer"
          ],
          "about": "A method for evaluating the quality of scientific news articles from their content, the scientific literature they reference and social media reactions to them. Data collection starts from Twitter postings on health and nutrition from June 2013 through June 2018, and indicators cover quotes, similarity to cited papers and the stance of social media reactions. Showing non-experts 7 indicators, such as site traffic, brought their ratings about 1 point (out of 5) closer to experts on 20 CRISPR articles, and slightly closer on 20 alcohol, tobacco and caffeine articles. An automated score from a weakly supervised classifier had the lowest error against expert ratings on CRISPR and tied for lowest on the other set.",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Outlet credibility",
            "Source referencing practices",
            "Trust indicators"
          ],
          "mdh_focus": [
            "M"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "Monitoring",
          "key_terms": [
            "Scientific literacy",
            "Health and nutrition",
            "Scientific news quality",
            "Social media stance",
            "Clickbait"
          ],
          "models": [
            "GloVe"
          ],
          "method_qualifiers": [
            "Weak supervision",
            "PageRank",
            "Bootstrapping",
            "Quote extraction"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Scientific-news stance-annotated tweet corpus",
              "kind": "dataset"
            },
            {
              "name": "SciLens",
              "kind": "framework"
            },
            {
              "name": "SciLens dataset",
              "kind": "dataset"
            }
          ],
          "data_description": "49K social media postings, 12K articles, 24K scientific links, 2013-2018.",
          "platform": [
            "Twitter"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Source referencing practices",
              "Outlet credibility",
              "Trust indicators"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "Outlet credibility",
              "Source referencing practices",
              "Trust indicators"
            ]
          },
          "tentative": false
        },
        {
          "title": "Stance Detection on Social Media with Fine-Tuned Large Language Models",
          "wid": "stance-detection-on-social-media-with-fine-tuned-large-",
          "type": "publication",
          "year": 2024,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2404.12171",
          "authors": [
            "Ilker Gül",
            "Rémi Lebret",
            "Karl Aberer"
          ],
          "epfl_authors": [
            "Ilker Gül",
            "Rémi Lebret",
            "Karl Aberer"
          ],
          "about": "A study fine-tuning the large language models ChatGPT, LLaMa-2 and Mistral-7B for stance detection on three public Twitter datasets, compared with baseline models on all three and with zero-shot and few-shot prompting on two. Fine-tuning outperformed zero-shot and few-shot prompting across all models, and fine-tuned ChatGPT reached an Favg of 79.7 on the Feminist Movement target of SemEval-2016, over 9 points above the top baseline. The larger fine-tuned LLaMa-2-13b did not consistently outperform the smaller LLaMa-2-7b or Mistral-7b, and both LLaMa-2 models showed near-peak performance on P-Stance with only 70 percent of the full dataset. Misinformation detection is named among the applications of stance detection.",
          "themes": [
            "NA"
          ],
          "subtopics": [],
          "mdh_focus": [
            "M"
          ],
          "mdh_relevance": "infrastructure",
          "relevance": 2,
          "stage": "Monitoring",
          "key_terms": [
            "Stance detection",
            "2020 US presidential election",
            "Fine-tuned LLMs",
            "Fine-tuning",
            "Large language models"
          ],
          "models": [
            "ChatGPT (gpt-3.5-turbo-1106)",
            "LLaMa-2-13b",
            "Mistral-7B"
          ],
          "method_qualifiers": [
            "LoRA",
            "Few-shot prompting"
          ],
          "events_cases": [
            "2020 US presidential election"
          ],
          "built_at_epfl": [],
          "data_description": "Three Twitter/X stance datasets: SemEval-2016, P-Stance, Twitter Stance Election 2020.",
          "platform": [
            "Twitter/X"
          ],
          "label_version": "v5",
          "theme_qualifiers": {},
          "theme_qualifiers_canonical": {},
          "tentative": false
        }
      ]
    },
    {
      "id": "faltings",
      "name": "Boi Faltings",
      "url": "https://people.epfl.ch/boi.faltings",
      "unit": "LIA",
      "faculty": "IC",
      "mdh_focus": [
        "M",
        "D"
      ],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "NLP",
        "LLMs",
        "LLM Alignment",
        "Argument Generation",
        "Preference Optimization"
      ],
      "stage": "Prevention",
      "publications": [
        {
          "title": "A Logical Fallacy-Informed Framework for Argument Generation",
          "wid": "a-logical-fallacy-informed-framework-for-argument-gener",
          "type": "publication",
          "year": 2025,
          "venue": "NAACL 2025 (Long Papers)",
          "link": "https://aclanthology.org/2025.naacl-long.374/",
          "authors": [
            "Luca Mouchel",
            "Debjit Paul",
            "Shaobo Cui",
            "Robert West",
            "Antoine Bosselut",
            "Boi Faltings"
          ],
          "epfl_authors": [
            "Luca Mouchel",
            "Debjit Paul",
            "Shaobo Cui",
            "Robert West",
            "Antoine Bosselut",
            "Boi Faltings"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "Logical fallacies",
            "Argument generation",
            "LLM alignment",
            "Preference optimization",
            "Misinformation prevention"
          ],
          "stage": "Prevention",
          "relevance": 2,
          "about": "Study treating the tendency of LLMs to produce arguments with logical fallacies as a misinformation risk, since flawed reasoning is harder for readers to catch than an outright false statement. It proposes a training method, FIPO, that adds a fallacy classification objective across 13 fallacy categories and weights the penalty by how common each fallacy is in real data, lowering fallacy rates to 17 percent for Llama-2 and 19.5 percent for Mistral and largely fixing the most stubborn category, faulty generalization. The authors flag the framework as dual use, since the same approach could generate more coherent but deliberately deceptive arguments.",
          "why": "It reduces fallacious LLM-generated arguments the authors link to spreading misinformation, while flagging a dual-use risk.",
          "data": "Text (ExplaGraphs; 7872 fallacious arguments across 13 fallacy categories)",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Fallacy-aware alignment"
          ],
          "key_terms": [
            "Logical fallacies",
            "Argument generation",
            "LLM alignment",
            "Preference optimization",
            "AI-generated misinformation risk"
          ],
          "models": [
            "ChatGPT",
            "ELECTRA",
            "GPT-4",
            "Llama-2",
            "Mistral"
          ],
          "method_qualifiers": [
            "LoRA",
            "LLM-as-a-judge",
            "Preference optimisation",
            "DPO"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Fallacy-augmented preference dataset",
              "kind": "dataset"
            },
            {
              "name": "FIPO (Fallacy-Informed Preference Optimization)",
              "kind": "framework"
            },
            {
              "name": "Fallacy preference dataset",
              "kind": "dataset",
              "url": "https://github.com/lucamouchel/Logical-Fallacies/tree/main/data",
              "evidence": "Paper footnote 1, page 1: \"Our code and datasets are publicly available for research purposes at github.com/lucamouchel/Logical-Fallacies\". The repository's data/ directory (opened) contains the folders argumentation/, generated/, preference-data/, sft/, sft_rag/ and the file test_debate.txt, matching the paper's pipeline (ExplaGraphs argumentation data plus ChatGPT-generated fallacious counterparts forming the preference pairs)."
            },
            {
              "name": "FIPO",
              "kind": "framework",
              "url": "https://github.com/lucamouchel/Logical-Fallacies",
              "evidence": "GitHub repository description reads \"A Logical Fallacy-Informed Framework for Argument Generation\"; the README credits Luca Mouchel, Debjit Paul, Shaobo Cui, Robert West, Antoine Bosselut and Boi Faltings, NAACL 2025, and the src/ tree carries the data collection, supervised fine-tuning, preference optimization (DPO, KTO, PPO, CPO) and FIPO stages described in the paper."
            }
          ],
          "follow_up": [
            {
              "what": "Official code and data release for the paper (GitHub: lucamouchel/Logical-Fallacies)",
              "kind": "repository",
              "url": "https://github.com/lucamouchel/Logical-Fallacies",
              "evidence": "Paper footnote 1: \"Our code and datasets are publicly available for research purposes at github.com/lucamouchel/Logical-Fallacies\" - confirmed by opening the repository, whose description is the paper title and which implements the FIPO training and evaluation pipeline."
            }
          ],
          "data_description": "7872 generated fallacious arguments over ExplaGraphs topics and stances.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "United States"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Fallacy-aware alignment"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Fallacy-aware alignment"
            ]
          },
          "tentative": false
        },
        {
          "title": "Unraveling Misinformation Propagation in LLM Reasoning",
          "wid": "unraveling-misinformation-propagation-in-llm-reasoning",
          "type": "publication",
          "year": 2025,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2505.18555",
          "authors": [
            "Yiyang Feng",
            "Yichen Wang",
            "Shaobo Cui",
            "Boi Faltings",
            "Mina Lee",
            "Jiawei Zhou"
          ],
          "epfl_authors": [
            "Yiyang Feng",
            "Shaobo Cui",
            "Boi Faltings"
          ],
          "about": "Study of how misinformation in user input, simulated as erroneous equations added to math questions, propagates through the reasoning of LLMs. Testing instruction-tuned and thinking models on 400 questions from MathQA, MATH, GSM8K and MetaMath, it finds that even when explicitly instructed to correct the misinformation, the models succeed less than half the time, with relative accuracy drops of 10.02 to 72.20 percent (4.30 to 19.97 percent for thinking models). Factual corrections applied early in the reasoning process are the most effective at reducing propagation, and fine-tuning on data with such corrections, combined with explicit correction instructions, raises GPT-4o-mini's accuracy under misinformation from 85.64 to 95.68 percent.",
          "themes": [
            "AI safety",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Equation fact-checking",
            "Error correction",
            "Factual robustness",
            "Hoax refutation",
            "Misuse risk assessment",
            "Safety alignment"
          ],
          "mdh_focus": [
            "M"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "Monitoring + Mitigation",
          "key_terms": [
            "Misinformation propagation",
            "Mathematical reasoning",
            "User-model knowledge conflicts",
            "Chain-of-thought reasoning",
            "LLM reasoning"
          ],
          "models": [
            "DeepSeek-R1-Distilled-Qwen-2.5",
            "GPT-4",
            "GPT-4o-mini",
            "Llama-3.2",
            "Mixtral",
            "Qwen-2",
            "Qwen-3"
          ],
          "method_qualifiers": [
            "Chain-of-thought prompting",
            "LoRA",
            "LLM-as-a-judge",
            "Fine-tuning"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "misinfo-prop",
              "kind": "dataset"
            },
            {
              "name": "Misinformation correction dataset",
              "kind": "dataset"
            }
          ],
          "data_description": "400 math questions across 4 datasets; 1054 fine-tuning correction pairs.",
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Factual robustness",
              "Safety alignment",
              "Misuse risk assessment"
            ],
            "Verification & content authenticity": [
              "Hoax refutation",
              "Error correction",
              "Equation fact-checking"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Factual robustness",
              "Misuse risk assessment",
              "Safety alignment"
            ],
            "Verification & content authenticity": [
              "Human fact-checking"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "vaudenay",
      "name": "Serge Vaudenay",
      "url": "https://people.epfl.ch/serge.vaudenay",
      "unit": "LASEC",
      "faculty": "IC",
      "mdh_focus": [
        "M",
        "D"
      ],
      "dataTypes": [
        "Data-agnostic",
        "Image"
      ],
      "techTypes": [
        "Cryptography",
        "Communication Security",
        "Biometrics",
        "Proof of Personhood",
        "Provenance Proofs"
      ],
      "stage": "Prevention",
      "publications": []
    },
    {
      "id": "donati",
      "name": "Laurène Donati & Mélissa Caloz",
      "url": "https://people.epfl.ch/laurene.donati",
      "unit": "Vice Presidency for Innovation and Impact (VPI)",
      "faculty": "EPFL",
      "mdh_focus": [],
      "dataTypes": [],
      "techTypes": [
        "Strategy",
        "Policy Engagement",
        "Media Relations",
        "Science Outreach"
      ],
      "stage": "Prevention",
      "publications": []
    },
    {
      "id": "zamir",
      "name": "Amir Zamir",
      "url": "https://people.epfl.ch/amir.zamir",
      "unit": "VILAB",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [
        "Image",
        "Video"
      ],
      "techTypes": [
        "Computer Vision",
        "Transfer Learning",
        "Visual Intelligence"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "kaser",
      "name": "Tanja Käser",
      "url": "https://people.epfl.ch/tanja.kaeser",
      "unit": "ML4ED",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "ML",
        "Educational AI",
        "AI Literacy"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "pthiran",
      "name": "Patrick Thiran",
      "url": "https://people.epfl.ch/patrick.thiran",
      "unit": "INDY",
      "faculty": "IC",
      "mdh_focus": [
        "M",
        "D"
      ],
      "dataTypes": [
        "Data-agnostic"
      ],
      "techTypes": [
        "Computational Social Science",
        "Graph Theory",
        "Network Source Localisation",
        "ML"
      ],
      "stage": "Monitoring + Mitigation",
      "publications": [
        {
          "title": "Reducing Sensor Requirements by Relaxing the Network Metric Dimension",
          "wid": "reducing-sensor-requirements-by-relaxing-the-network-me",
          "type": "publication",
          "year": 2025,
          "venue": "ACM SIGMETRICS 2025 / POMACS",
          "link": "https://arxiv.org/abs/2505.11193",
          "authors": [
            "Paula Mürmann",
            "Robin Jaccard",
            "Maximilien Dreveton",
            "Aryan Alavi Razavi Ravari",
            "Patrick Thiran"
          ],
          "epfl_authors": [
            "Paula Mürmann",
            "Robin Jaccard",
            "Maximilien Dreveton",
            "Aryan Alavi Razavi Ravari",
            "Patrick Thiran"
          ],
          "mdh_relevance": "infrastructure",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "Source localisation",
            "Graph theory",
            "Sensor placement",
            "Network monitoring",
            "Misinformation source detection"
          ],
          "stage": "NA",
          "relevance": 2,
          "about": "Study of source localization on graphs: given that something such as an epidemic, a piece of misinformation, or a cyber-attack started at an unknown node and spread, it recovers the origin from distance measurements at a small set of sensor nodes. Because the classical metric dimension requires too many sensors, the paper introduces a k-relaxed metric dimension that only needs nodes more than k steps apart to be distinguishable. It proves results for trees, gives a greedy algorithm for general graphs, and proposes a two-step strategy that does coarse localization with few sensors then refines.",
          "why": "It lists misinformation source identification as a direct application, locating the origin of a rumour or cascade without instrumenting every node.",
          "data": "Graph/Network (random trees, random geometric graphs, real-world networks)",
          "themes": [
            "NA"
          ],
          "subtopics": [],
          "key_terms": [
            "Source localization",
            "Metric dimension",
            "Sensor placement",
            "Random trees",
            "Epidemic source detection"
          ],
          "models": [],
          "method_qualifiers": [
            "Greedy approximation algorithm",
            "Random-tree asymptotics",
            "Simulation-based evaluation"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "follow_up": [
            {
              "what": "metric-dimension-relaxed, the simulation code released for this paper",
              "kind": "repository",
              "url": "https://github.com/mdreveton/metric-dimension-relaxed",
              "evidence": "The paper's own Code availability statement reads: 'Code availability The code to reproduce the simulations is available at https://github.com/mdreveton/metric-dimension-relaxed/'. The repository README confirms the match: 'Code to reproduce the results of the paper \"Reducing Sensor Requirements by Relaxing the Network Metric Dimension\"', listing 'experiments.py: file to run the experiments of Section 4', 'sequentialGame.py: file to run the two step localization game of Section 5', and a Figure folder holding 'the figures of the paper + some additional figures'. The account owner mdreveton is co-author Maximilien Dreveton (EPFL)."
            }
          ],
          "data_description": "Random trees, random geometric graphs and four real-world networks.",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {},
          "theme_qualifiers_canonical": {},
          "tentative": false
        },
        {
          "title": "A General Framework for Sensor Placement in Source Localization",
          "wid": "a-general-framework-for-sensor-placement-in-source-loca",
          "type": "publication",
          "year": 2019,
          "venue": "IEEE Transactions on Network Science and Engineering 6(2)",
          "link": "https://doi.org/10.1109/TNSE.2017.2787551",
          "authors": [
            "Brunella Spinelli",
            "L. Elisa Celis",
            "Patrick Thiran"
          ],
          "epfl_authors": [
            "Brunella Spinelli",
            "L. Elisa Celis",
            "Patrick Thiran"
          ],
          "about": "Framework for finding an epidemic's source in a network, such as a rumor or disease outbreak, from infection states and times revealed by a small set of sensor nodes. It uses static sensors alone, placed independently of any particular epidemic, or adds dynamic sensors as the epidemic spreads, and localizes the source after the epidemic has spread through the entire network or while it is ongoing. Experiments on synthetic and real-world networks show dynamic sensors cut the number of sensors needed by up to a factor of 10 and, even with high-variance transmission delays, the source can be localized with fewer than 5 percent of nodes as sensors.",
          "themes": [
            "NA"
          ],
          "subtopics": [],
          "mdh_focus": [
            "M"
          ],
          "mdh_relevance": "infrastructure",
          "relevance": 2,
          "stage": "Monitoring",
          "key_terms": [
            "Sensor placement",
            "Rumor source detection",
            "Source localization",
            "Rumor propagation",
            "Epidemic spreading"
          ],
          "models": [],
          "method_qualifiers": [
            "Greedy approximation algorithm",
            "Adaptive sensor placement",
            "Epidemic source inference",
            "Simulation-based evaluation"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "Synthetic and real-world networks (Facebook, Airline), up to 3732 nodes.",
          "label_version": "v5",
          "theme_qualifiers": {},
          "theme_qualifiers_canonical": {},
          "tentative": false
        },
        {
          "title": "Back to the Source: An Online Approach for Sensor Placement and Source Localization",
          "wid": "back-to-the-source-an-online-approach-for-sensor-placem",
          "type": "publication",
          "year": 2017,
          "venue": "arXiv preprint",
          "link": "https://infoscience.epfl.ch/handle/20.500.14299/136214",
          "authors": [
            "Brunella Spinelli",
            "L. Elisa Celis",
            "Patrick Thiran"
          ],
          "epfl_authors": [
            "Brunella Spinelli",
            "L. Elisa Celis",
            "Patrick Thiran"
          ],
          "about": "Method for source localization, finding the originator of a disease or rumor in a network, that places sensors online rather than all in advance. A small number of static sensors detects the epidemic, then the method iteratively chooses the most informative node as a new sensor, so the source can be found while the epidemic is still ongoing. It applies to general network topologies and random transmission delays. On synthetic networks with a limited budget, it raises the success rate over an equal-budget static strategy from about 5 to about 75 percent, and with no budget and deterministic delays uses about 3 percent of nodes as sensors.",
          "themes": [
            "NA"
          ],
          "subtopics": [],
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_relevance": "infrastructure",
          "relevance": 2,
          "stage": "Monitoring",
          "key_terms": [
            "Online sensor placement",
            "Rumor spreading",
            "Source localization",
            "Sensor placement",
            "Diffusion networks"
          ],
          "models": [],
          "method_qualifiers": [
            "Greedy approximation algorithm",
            "Source localization",
            "Online sensor placement"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "Synthetic and real-world networks (Facebook, Airline), 100 simulations per topology.",
          "label_version": "v5",
          "theme_qualifiers": {},
          "theme_qualifiers_canonical": {},
          "tentative": false
        }
      ]
    },
    {
      "id": "rajman",
      "name": "Martin Rajman",
      "url": "https://people.epfl.ch/martin.rajman",
      "unit": "AI Center / SNAI",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "NLP",
        "Information Extraction",
        "Text Mining"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "payer",
      "name": "Mathias Payer",
      "url": "https://people.epfl.ch/mathias.payer",
      "unit": "HexHive",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [],
      "techTypes": [
        "Security"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "aad_milliquet",
      "name": "Jean-Pierre Hubaux, Imad Aad & Stéphanie Milliquet",
      "unit": "C4DT",
      "faculty": "IC",
      "dataTypes": [],
      "techTypes": [
        "Security",
        "Privacy Engineering",
        "Tech Transfer"
      ],
      "stage": "Prevention",
      "publications": [
        {
          "title": "Digital Dilemmas: Humanitarian Consequences (exhibition)",
          "wid": "digital-dilemmas-humanitarian-consequences-exhibitio",
          "type": "event",
          "year": 2024,
          "venue": "EPFL Pavilions, 3 May to 14 July 2024. EPFL EssentialTech Centre with the ICRC, co-organised with EPFL Pavilions, in partnership with the Center for Digital Trust (C4DT)",
          "link": "https://epfl-pavilions.ch/fr/exhibitions/digital-dilemmas",
          "authors": [
            "Grégoire Castella (EssentialTech)",
            "Louis Potter (EssentialTech)",
            "Klaus Schönenberger (EssentialTech)",
            "Sarah Kenderdine (EPFL Pavilions)",
            "Marie Carrard (EPFL Pavilions)",
            "Peter Grönquist (IVRL)",
            "Stéphanie Milliquet (C4DT)",
            "Philippe Stoll (ICRC)",
            "Catherina Zazzini (ICRC)",
            "Fabrice Lauper"
          ],
          "epfl_authors": [
            "Grégoire Castella (EssentialTech)",
            "Louis Potter (EssentialTech)",
            "Klaus Schönenberger (EssentialTech)",
            "Sarah Kenderdine (EPFL Pavilions)",
            "Marie Carrard (EPFL Pavilions)",
            "Peter Grönquist (IVRL)",
            "Stéphanie Milliquet (C4DT)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_topics": [
            "Public exhibition",
            "Digital ethics",
            "Awareness",
            "ICRC"
          ],
          "stage": "Prevention",
          "relevance": 3,
          "about": "Immersive public exhibition at EPFL Pavilions exploring the digital risks that civilian populations and humanitarian workers face in conflict zones, where digital tools open access to essential services but can also expose personal data, enable surveillance or worsen misinformation. Visitors worked through dilemmas covering biometrics, civilian involvement in digital warfare, misinformation, AI-generated deepfakes, algorithmic decision-making, connectivity and data protection, and is followed by an installation of solutions developed by EPFL laboratories with the ICRC and ETH Zurich. It first appeared at UN headquarters in New York before this expanded Swiss presentation.",
          "why": "It addresses misinformation and AI deepfakes in humanitarian settings, raising public awareness.",
          "data": "not applicable (public exhibition)",
          "lab": [
            "EssentialTech Centre",
            "IVRL",
            "EPFL Pavilions",
            "C4DT",
            "ICRC"
          ],
          "themes": [
            "Media literacy & public resilience"
          ],
          "subtopics": [
            "Digital literacy",
            "Immersive demonstration",
            "Public awareness campaigns"
          ],
          "key_terms": [
            "Humanitarian action",
            "Deepfakes",
            "Hate speech",
            "Conflict zones",
            "Armed conflict"
          ],
          "models": [],
          "method_qualifiers": [],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Digital Dilemmas: Humanitarian Consequences",
              "kind": "platform"
            }
          ],
          "follow_up": [
            {
              "what": "Digital Dilemmas 2.0 - the \"Deepfake and You\" exhibit, built by the ICRC with EPFL's Image and Visual Representations Lab and shown in the foyer of the UN General Assembly building in New York",
              "kind": "successor-work",
              "url": "https://blogs.icrc.org/intercross/2025/03/20/digital-dilemmas-2-0-deepfakes/",
              "evidence": "ICRC Intercross podcast page, published 20 March 2025, titled \"Digital Dilemmas 2.0: Deepfakes\": \"We tour a deepfake exhibit created by the ICRC and L'Ecole Polytechnique Federale de Lausanne (EPFL) at the United Nations.\" In the transcript, EPFL research software engineer Peter Gronquist says: \"So we're at the UN right now. And I'm about to give you a short tour of this deep fake exhibit that we've been setting up with ICRC and EPFL.\" The page adds: \"The ICRC is working with EPFL to explain and find solutions to deepfakes and other technologies disseminating harmful information among populations in conflict areas.\""
            }
          ],
          "data_description": "No empirical data (exhibition description).",
          "follow_up_checked": "2026-09-02",
          "targeted_group": [
            "Civilian populations",
            "Civilians in conflict zones",
            "Humanitarian aid organisations",
            "Humanitarian aid workers"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Media literacy & public resilience": [
              "Public awareness campaigns",
              "Immersive demonstration",
              "Digital literacy"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media literacy & public resilience": [
              "Digital literacy",
              "Public awareness campaigns"
            ]
          },
          "tentative": true
        }
      ]
    },
    {
      "id": "overdorf",
      "name": "Rebekah Overdorf",
      "unit": "LSIR",
      "faculty": "External (Ruhr University Bochum)",
      "dataTypes": [
        "Text",
        "Data-agnostic"
      ],
      "techTypes": [
        "Security",
        "Privacy"
      ],
      "stage": "Monitoring + Mitigation",
      "publications": [
        {
          "title": "Disinformation from the Inside, Combining Machine Learning and Journalism to Investigate Sockpuppet Campaigns",
          "wid": "disinformation-from-the-inside-combining-machine-learni",
          "type": "publication",
          "year": 2020,
          "link": "https://doi.org/10.1145/3366424.3385777",
          "authors": [
            "Christopher Schwartz (KU Leuven)",
            "Rebekah Overdorf (EPFL LSIR)"
          ],
          "epfl_authors": [
            "Rebekah Overdorf (EPFL LSIR)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Sockpuppet campaigns",
            "Disinformation typology"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "Study combining machine learning with investigative journalism to examine a sockpuppet disinformation campaign in Kyrgyzstan, with tweet collection oriented to the October 2017 presidential election. Sockpuppets are human-controlled fake accounts, and the paper argues that as bot detection improves, adversaries shift toward them, especially infiltrators that integrate into a target community to persuade it from within. It sets out the infiltrator as a subset of sockpuppets, set against bots, whose chief effect the authors describe as driving traffic or drowning out opposition while infiltrators assimilate genuine audiences from within. The Kyrgyz case draws on a whistleblower who described the sockpuppets as \"high quality\", which the authors gloss as few in number but resource-intensive; separately the authors estimate that the average size of a sockpuppet campaign in Kyrgyzstan ranges between 10 and 100 fake accounts, and the typology grounds later EPFL work on astroturfing and Telegram propaganda accounts.",
          "why": "It names the infiltrator as an adversarial escalation of meatpuppets and Sybils, and documents a real sockpuppet disinformation campaign.",
          "data": "Network",
          "themes": [
            "Influence operations & coordinated manipulation",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Sockpuppet accounts"
          ],
          "key_terms": [
            "Kyrgyzstan",
            "Investigative journalism",
            "Sockpuppet campaigns",
            "Infiltrator accounts",
            "Bot-detection evasion"
          ],
          "models": [],
          "method_qualifiers": [
            "Journalist-in-the-loop supervision",
            "Supervised classification",
            "Manual annotation",
            "Adversarial evasion"
          ],
          "events_cases": [
            "2017 Kyrgyz presidential election",
            "Kyrgyzstan 2017 presidential election",
            "Kyrgyzstan sockpuppet disinformation campaigns",
            "RE:AKIA anti-corruption protests"
          ],
          "built_at_epfl": [],
          "data_description": "Facebook and Twitter accounts from Kyrgyzstan, manually annotated; one whistleblower source.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Facebook",
            "Twitter/X"
          ],
          "region_country": [
            "Kyrgyzstan"
          ],
          "targeted_group": [
            "Ethnic minorities",
            "Political opposition"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Influence operations & coordinated manipulation": [
              "Sockpuppet accounts"
            ]
          },
          "theme_qualifiers_canonical": {
            "Influence operations & coordinated manipulation": [
              "Sockpuppet accounts"
            ]
          },
          "tentative": true
        },
        {
          "title": "Tactical Reframing of Online Disinformation Campaigns Against The Istanbul Convention",
          "wid": "tactical-reframing-of-online-disinformation-campaigns-a",
          "type": "publication",
          "year": 2021,
          "link": "https://arxiv.org/abs/2105.13398",
          "authors": [
            "Tugrulcan Elmas (LSIR)",
            "Rebekah Overdorf (EPFL)",
            "Karl Aberer (LSIR)"
          ],
          "epfl_authors": [
            "Tugrulcan Elmas (LSIR)",
            "Rebekah Overdorf (EPFL)",
            "Karl Aberer (LSIR)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D",
            "H"
          ],
          "mdh_topics": [
            "Disinformation campaigns",
            "Narrative reframing",
            "Homophobia",
            "Gender-based violence",
            "Turkey",
            "Facebook",
            "Tactical reframing",
            "Astroturfing",
            "Cross-actor coordination"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "Empirical study tracing how an online disinformation campaign in Turkey shifted its message to build support for leaving the Istanbul Convention, the human-rights treaty on violence against women. Using public Facebook posts, it shows the campaign began in divorced men's groups complaining about the domestic implementing law, then was reframed to attack the convention by stressing its recognition of sexual orientation and non-traditional gender roles. Small men's-rights groups had their content amplified by larger political and religious pages and by a pro-government newspaper that shifted its own coverage the same way. It is presented as the first case study of narrative reframing inside a social-media disinformation campaign.",
          "why": "It documents a disinformation campaign that fused false framing with homophobic hate to roll back women's rights.",
          "data": "Text",
          "themes": [
            "Gender-based violence & misogyny",
            "Influence operations & coordinated manipulation",
            "Media framing & narrative analysis"
          ],
          "subtopics": [
            "Anti-gender campaigns",
            "Men's rights groups",
            "Narrative reframing",
            "Weaponised homophobia"
          ],
          "key_terms": [
            "Istanbul Convention",
            "Tactical reframing",
            "Homophobia",
            "Men's rights groups",
            "Turkey"
          ],
          "models": [],
          "method_qualifiers": [
            "Frame analysis",
            "Manual annotation",
            "Retrospective archive mining"
          ],
          "events_cases": [
            "Turkey's withdrawal from the Istanbul Convention"
          ],
          "built_at_epfl": [],
          "data_description": "Facebook posts from 2500 tracked Turkish groups, pages and profiles, 2014-2020.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Facebook"
          ],
          "region_country": [
            "Turkey"
          ],
          "targeted_group": [
            "LGBTQ+",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Gender-based violence & misogyny": [
              "Weaponised homophobia",
              "Anti-gender campaigns",
              "Men's rights groups"
            ],
            "Influence operations & coordinated manipulation": [
              "Grassroots campaign organizing",
              "Narrative reframing"
            ],
            "Media framing & narrative analysis": [
              "Narrative shift over time"
            ]
          },
          "theme_qualifiers_canonical": {
            "Gender-based violence & misogyny": [
              "Anti-gender campaigns",
              "Men's rights groups",
              "Weaponised homophobia"
            ],
            "Influence operations & coordinated manipulation": [
              "Grassroots campaign organizing",
              "Narrative reframing"
            ],
            "Media framing & narrative analysis": [
              "Issue framing",
              "Narrative shift over time"
            ]
          },
          "tentative": false
        },
        {
          "title": "Characterizing and Detecting Propaganda-Spreading Accounts on Telegram",
          "wid": "characterizing-and-detecting-propaganda-spreading-accou",
          "type": "publication",
          "year": 2025,
          "link": "https://www.usenix.org/conference/usenixsecurity25/presentation/kireev",
          "authors": [
            "Klim Kireev (EPFL / MPI-SP)",
            "Yevhen Mykhno (independent)",
            "Carmela Troncoso (EPFL / MPI-SP)",
            "Rebekah Overdorf (UNIL / Ruhr University Bochum)"
          ],
          "epfl_authors": [
            "Klim Kireev (EPFL / MPI-SP)",
            "Carmela Troncoso (EPFL / MPI-SP)",
            "Rebekah Overdorf (UNIL / Ruhr University Bochum)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Propaganda",
            "Telegram",
            "Coordinated accounts",
            "Russia-Ukraine",
            "Detection",
            "Content moderation",
            "Telegram propaganda detection",
            "Russian information operations"
          ],
          "stage": "Monitoring + Mitigation",
          "relevance": 5,
          "about": "Study of how propaganda operates on Telegram, a platform where messages appear chronologically and moderation is left to channel owners. It builds a new labeled dataset of messages from political and news channels in Russian, Belarusian and Ukrainian, combining a long historical export with real-time capture so deleted messages are preserved. It surfaces two independent coordinated networks, one pro-Russian and one pro-Ukrainian, whose propaganda messages drew about as many replies as messages from real users. Using only the information a moderator can see, it builds a classifier that detects propaganda from a single observed message, reaching 97.4 percent accuracy and holding up on unseen topics. It received a Distinguished Paper Award at USENIX Security 2025.",
          "why": "It maps two coordinated disinformation networks and delivers a propaganda detector.",
          "data": "Text, 17.3M messages from 13 political/news channels (Russian, Belarusian, Ukrainian)",
          "themes": [
            "Content moderation & enforcement",
            "Influence operations & coordinated manipulation",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Account coordination graphs",
            "Channel-level moderation",
            "Coordinated propaganda networks",
            "Moderator-assist detection",
            "Tactical reframing"
          ],
          "key_terms": [
            "Russo-Ukrainian war",
            "Propaganda accounts",
            "Instant-messaging platforms",
            "Propaganda detection",
            "Coordinated account networks"
          ],
          "models": [
            "GPT-4",
            "SBERT",
            "XGBoost"
          ],
          "method_qualifiers": [
            "Adversarial evasion",
            "Supervised classification"
          ],
          "events_cases": [
            "Russo-Ukrainian war",
            "Wagner Group rebellion"
          ],
          "built_at_epfl": [
            {
              "name": "Telegram propaganda dataset",
              "kind": "dataset",
              "url": "https://zenodo.org/records/14736756",
              "evidence": "Paper, Contributions: \"We compile the first labeled Telegram propaganda dataset of group messages and channel comments. This dataset comprises 17.3M labeled messages from 13 political and news-oriented channels. ... The dataset is available on Zenodo 4.\" and, in the Data availability statement: \"Our dataset is published on Zenodo 4.\" The footnote prints \"https://zenodo.org/records/14736756\". The Zenodo record itself is titled \"Real-time and Historical Telegram Dataset Annotated by Propaganda\", authored by Kireev, Klim (Ecole Polytechnique Federale de Lausanne), with Mykhno, Yevgen as contributor and Troncoso, Carmela and Overdorf, Rebekah as supervisors, licensed CC BY 4.0."
            }
          ],
          "follow_up": [
            {
              "what": "A Telegram Dataset of Propaganda and its Moderation, ICWSM 2025 (Proceedings of the International AAAI Conference on Web and Social Media, vol. 19 no. 1, pp. 2510-2518) - a dedicated dataset paper by the same four authors that publishes and documents the data collected for this work",
              "kind": "successor-work",
              "url": "https://ojs.aaai.org/index.php/ICWSM/article/view/35952",
              "evidence": "ICWSM paper, Dataset section: \"we identified in our prior work (Kireev et al. 2024)\"; Channel Selection: \"Second, smaller channels that we identified manually as having apparently automated propaganda activity as described in our prior work (Kireev et al. 2024).\"; Discussion: \"As we explored in prior work (Kireev et al. 2024), this dataset can be used to detect and remove propaganda from Telegram. We implemented a classifier to detect future propaganda based on the needs and abilities of channel owners.\" Its reference list resolves that citation as \"Kireev, K.; Mykhno, Y.; Troncoso, C.; and Overdorf, R. 2024. Characterizing and Detecting Propaganda-Spreading Accounts on Telegram. arXiv preprint arXiv:2406.08084.\", i.e. this work."
            }
          ],
          "data_description": "17.3M labeled Telegram messages from 13 channels, collected 2023.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Telegram"
          ],
          "region_country": [
            "Russia",
            "Ukraine"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Channel-level moderation",
              "Moderator-assist detection"
            ],
            "Influence operations & coordinated manipulation": [
              "Coordinated propaganda networks",
              "Inauthentic account behaviour",
              "Tactical reframing"
            ],
            "Spread, amplification & networks": [
              "Account coordination graphs"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Channel-level moderation",
              "Moderator-assist detection"
            ],
            "Influence operations & coordinated manipulation": [
              "Coordinated propaganda networks",
              "Narrative reframing",
              "Sockpuppet accounts"
            ],
            "Spread, amplification & networks": [
              "Account coordination graphs",
              "Cross-platform migration"
            ]
          },
          "tentative": false
        },
        {
          "title": "Vaccine-disinformation project (current case study)",
          "wid": "vaccine-disinformation-project-current-case-study",
          "type": "project",
          "year": 2026,
          "venue": "Overdorf lab",
          "authors": [
            "Rebekah Overdorf"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Vaccine disinformation",
            "Unambiguous-case framing",
            "Health misinformation"
          ],
          "stage": "NA",
          "relevance": 4,
          "about": "Ongoing case study of disinformation moderation that deliberately picks cases where the truth is not in dispute, such as the claim that vaccines cause autism, climate denial, and Holocaust denial. Because these claims are easy to label and classify, the moderation problem becomes tractable and the analysis can focus on amplification and enforcement rather than adjudicating contested facts. It fits a broader approach of measuring manipulation, coordination, and platform-level moderation effects.",
          "why": "It studies the moderation of clear-cut health and conspiracy disinformation.",
          "data": "Text, vaccine/climate/Holocaust denial content",
          "lab": [],
          "themes": [
            "Content moderation & enforcement",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Bot amplification",
            "Coordinated inauthentic behaviour",
            "Coordination measurement",
            "Moderation effectiveness",
            "Uncontested-truth case selection"
          ],
          "key_terms": [
            "Vaccine disinformation",
            "Climate denial",
            "Holocaust denial",
            "Content moderation",
            "Amplification"
          ],
          "models": [],
          "method_qualifiers": [],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "Vaccine/climate/Holocaust-denial text content.",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Moderation effectiveness",
              "Uncontested-truth case selection"
            ],
            "Influence operations & coordinated manipulation": [
              "Bot amplification",
              "Coordination measurement",
              "Coordinated inauthentic behaviour"
            ],
            "Spread, amplification & networks": [
              "Amplification analysis"
            ]
          },
          "theme_qualifiers_canonical": {
            "Influence operations & coordinated manipulation": [
              "Bot amplification"
            ],
            "Spread, amplification & networks": [
              "Account coordination graphs"
            ]
          },
          "tentative": true
        },
        {
          "title": "It is not enough to give your moderation rules to ChatGPT: Policy-as-Prompt Moderation and Its Potential Impacts on Community Governance",
          "wid": "it-is-not-enough-to-give-your-moderation-rules-to-chatg",
          "type": "publication",
          "year": 2026,
          "venue": "Mensch und Computer 2026",
          "link": "https://arxiv.org/abs/2607.12149",
          "authors": [
            "Anna Neumann",
            "Jasmin Wyss",
            "Ivy Turk",
            "Rebekah Overdorf"
          ],
          "about": "Written at Ruhr University Bochum, after Rebekah Overdorf moved there from the University of Lausanne in 2025. Analysis of the policy-as-prompt approach to content moderation, where a moderation policy is formulated as a natural-language prompt and passed to a large language model (LLM) that aids in moderation tasks. It sets out the approach's technical and governance properties, such as prompt injections and prompt stacks that prioritize instructions set by foundation-model developers, and traces its possible impacts on centralized moderation and on decentralized moderation, where community members could lose vital expertise. The authors find that writing prompts alone is not appropriate for ensuring meaningful community governance and argue that, as LLMs currently cannot take accountability for their moderation actions, they should only assist human decisions or supplement human deliberation.",
          "themes": [
            "Content moderation & enforcement",
            "Platform governance & regulation"
          ],
          "subtopics": [
            "Human accountability",
            "Moderator-assist detection",
            "Prompt governance",
            "Transparency obligations"
          ],
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "Mitigation",
          "key_terms": [
            "Policy-as-prompt",
            "Prompt governance",
            "Community self-governance",
            "System prompt hierarchy",
            "Content moderation"
          ],
          "models": [],
          "method_qualifiers": [
            "Conceptual policy analysis"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "No empirical data (policy analysis).",
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Policy-as-prompt moderation",
              "Moderator-assist detection",
              "Human accountability"
            ],
            "Platform governance & regulation": [
              "Prompt governance",
              "Transparency obligations",
              "Moderation accountability"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Moderator-assist detection"
            ],
            "Platform governance & regulation": [
              "Transparency obligations"
            ]
          },
          "tentative": false
        },
        {
          "title": "Subtle Censorship via Adversarial Fakeness in Kyrgyzstan",
          "wid": "subtle-censorship-via-adversarial-fakeness-in-kyrgyzsta",
          "type": "publication",
          "year": 2019,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/1906.08021",
          "authors": [
            "Christopher Schwartz",
            "Rebekah Overdorf"
          ],
          "epfl_authors": [
            "Rebekah Overdorf"
          ],
          "about": "Talk using Kyrgyzstan as its main case study to examine how fake news and fake profiles act as subtle censorship, which relies on pretence and imitation and obscures whether censorship is happening. It characterises fakeness by adversarialness, the active intention to falsify or mislead to benefit the agent and harm the target, and describes silencing through attacks on journalists and activists, through noise, and through misinformation, as when fake news sparked the June 2010 inter-ethnic clashes between Kyrgyz and Uzbeks. The authors' project combines machine learning, philosophy and journalism, manually identifying fake accounts that seed fake news and the people who re-propagate it before training classifiers on social graph features.",
          "themes": [
            "Influence operations & coordinated manipulation",
            "Spread, amplification & networks",
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Hashtag poisoning",
            "Identity leasing",
            "Information cascades",
            "Journalist intimidation",
            "Popularity mechanism manipulation",
            "Sockpuppet accounts"
          ],
          "mdh_focus": [
            "D",
            "H"
          ],
          "mdh_relevance": "direct",
          "relevance": 5,
          "stage": "Monitoring",
          "key_terms": [
            "Subtle censorship",
            "Adversarial fakeness",
            "Fake profiles",
            "Journalist intimidation",
            "Fake news"
          ],
          "models": [],
          "method_qualifiers": [
            "Manual annotation",
            "Supervised classification",
            "Journalist-in-the-loop supervision"
          ],
          "events_cases": [
            "2010 Kyrgyzstan inter-ethnic clashes",
            "2018 anti-Chinese demonstrations in Kyrgyzstan",
            "Nedim Turfent case"
          ],
          "built_at_epfl": [],
          "data_description": "Planned crawl of fake-news posts and fake-account profiles in Kyrgyzstan.",
          "platform": [
            "Facebook",
            "Twitter/X"
          ],
          "region_country": [
            "Kyrgyzstan",
            "Mexico",
            "Turkey"
          ],
          "targeted_group": [
            "Activists",
            "Chinese",
            "Ethnic Uzbeks",
            "Journalists",
            "Opposition politicians",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Influence operations & coordinated manipulation": [
              "Sockpuppet accounts",
              "Hashtag poisoning",
              "Identity leasing"
            ],
            "Spread, amplification & networks": [
              "Popularity mechanism manipulation",
              "Information cascades"
            ],
            "Toxicity & harassment": [
              "Journalist intimidation"
            ]
          },
          "theme_qualifiers_canonical": {
            "Influence operations & coordinated manipulation": [
              "Sockpuppet accounts"
            ],
            "Spread, amplification & networks": [
              "Influential-user tracking",
              "Information cascades",
              "Popularity mechanism manipulation"
            ],
            "Toxicity & harassment": [
              "Online harassment of women"
            ]
          },
          "tentative": false
        },
        {
          "title": "Thinking Taxonomically about Fake Accounts: Classification, False Dichotomies, and the Need for Nuance",
          "wid": "thinking-taxonomically-about-fake-accounts-classificati",
          "type": "publication",
          "year": 2020,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2006.04959",
          "authors": [
            "Rebekah Overdorf",
            "Christopher Schwartz"
          ],
          "epfl_authors": [
            "Rebekah Overdorf"
          ],
          "about": "Paper proposing a systematic way to think taxonomically about fake accounts, a primary vector for misinformation and disinformation, arguing that research has used machine learning classifiers as a substitute for taxonomies, leading to binary conclusions and false dichotomies. Using a thought experiment from the novel Ender's Game, it argues that Facebook's Coordinated Inauthentic Behavior framework obscures the role of intention, which a correct classifier would need as an input but is not observable on platforms. It deconstructs four false dichotomies (coordination versus non-coordination, program versus person, deception versus forthrightness, inauthenticity versus authenticity) and proposes four matching aspects: scale, users, purposes and techniques, and audience impacts and implications.",
          "themes": [
            "Influence operations & coordinated manipulation",
            "Platform governance & regulation"
          ],
          "subtopics": [
            "Bot amplification",
            "Conceptual framework",
            "Deceptive intent attribution",
            "Online anonymity",
            "Sockpuppet accounts",
            "Technical standards"
          ],
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "NA",
          "key_terms": [
            "Fake accounts",
            "Coordinated inauthentic behavior",
            "Sockpuppets",
            "Coordinated Inauthentic Behavior",
            "Taxonomy"
          ],
          "models": [],
          "method_qualifiers": [
            "Philosophical analysis",
            "Conceptual argument",
            "Thought experiments"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "No empirical data (conceptual analysis).",
          "platform": [
            "Facebook",
            "Twitter/X"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Influence operations & coordinated manipulation": [
              "Sockpuppet accounts",
              "Bot amplification",
              "Deceptive intent attribution"
            ],
            "Platform governance & regulation": [
              "Online anonymity",
              "Technical standards",
              "Conceptual framework"
            ]
          },
          "theme_qualifiers_canonical": {
            "Influence operations & coordinated manipulation": [
              "Bot amplification",
              "Sockpuppet accounts"
            ],
            "Platform governance & regulation": [
              "Online anonymity",
              "Technical standards"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "shrestha",
      "name": "Yash Raj Shrestha & Amirsiavosh Bashardoust",
      "unit": "Applied Artificial Intelligence Lab",
      "faculty": "External (UNIL HEC)",
      "dataTypes": [
        "Text",
        "Image"
      ],
      "techTypes": [
        "ML",
        "NLP",
        "HCI",
        "Local LLMs"
      ],
      "stage": "Prevention + Monitoring",
      "publications": [
        {
          "title": "Comparing the Willingness to Share for Human-generated vs. AI-generated Fake News",
          "wid": "comparing-the-willingness-to-share-for-human-generated-",
          "type": "publication",
          "year": 2024,
          "venue": "Proceedings of the ACM on Human-Computer Interaction 8, CSCW2, 1-21",
          "link": "https://dl.acm.org/doi/10.1145/3687028",
          "authors": [
            "Amirsiavosh Bashardoust (UNIL)",
            "Stefan Feuerriegel (LMU Munich)",
            "Yash Raj Shrestha (UNIL)"
          ],
          "epfl_authors": [
            "Amirsiavosh Bashardoust (UNIL)",
            "Yash Raj Shrestha (UNIL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "AI-generated misinformation",
            "User behaviour"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "User study comparing how people react to AI-generated versus human-written fake news. In a pre-registered online experiment rating COVID-19 fake-news items, participants judged the AI-generated versions as less accurate yet were just as willing to share both. Lower perceived credibility did not translate into less sharing, so AI-authored falsehoods spread about as readily as human-made ones.",
          "why": "It measures real sharing behaviour for AI-generated misinformation.",
          "data": "Text, user-study responses",
          "themes": [
            "Persuasion & cognitive effects",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Audience susceptibility",
            "LLM persuasion",
            "Perceived veracity",
            "Self-amplification",
            "Sharing propensity"
          ],
          "key_terms": [
            "AI-generated fake news",
            "COVID-19 misinformation",
            "Willingness to share",
            "Perceived veracity",
            "COVID-19 fake news"
          ],
          "models": [
            "GPT-4"
          ],
          "method_qualifiers": [
            "Within-subject experiment",
            "Mixed-effects regression",
            "Preregistered design"
          ],
          "events_cases": [
            "COVID-19 pandemic"
          ],
          "built_at_epfl": [
            {
              "name": "Human-generated vs. AI-generated COVID-19 fake news corpus",
              "kind": "dataset"
            }
          ],
          "follow_up": [
            {
              "what": "Code and data release for the experiment: raw and processed survey exports, the preprocessing notebook, the R analysis, and a PDF of the fake news items shown to subjects",
              "kind": "repository",
              "url": "https://github.com/vosh-96/Comparing-the-willingness-to-share-for-human-generated-vs.-AI-generated-fake-news",
              "evidence": "Code and data supporting our findings are available at this public Git repository https://github.com/vosh-96/Comparing-the-willingness-to-share-for-human-generated-vs.-AI-generated-fake-news."
            },
            {
              "what": "OSF project holding the pre-registered experimental protocol, contributed by Amirsiavosh Bashardoust and Stefan Feuerriegel",
              "kind": "repository",
              "url": "https://osf.io/96sjx/",
              "evidence": "The pre-registration protocol is available at https://osf.io/96sjx/."
            }
          ],
          "data_description": "988 US subjects, 20 fake news items (10 human, 10 AI), COVID-19 topic.",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "Persuasion & cognitive effects": [
              "Audience susceptibility",
              "Perceived veracity",
              "LLM persuasion"
            ],
            "Spread, amplification & networks": [
              "Sharing propensity",
              "Self-amplification"
            ]
          },
          "theme_qualifiers_canonical": {
            "Persuasion & cognitive effects": [
              "Attitude change",
              "Audience susceptibility",
              "LLM persuasion"
            ],
            "Spread, amplification & networks": [
              "Popularity mechanism manipulation",
              "Sharing propensity"
            ]
          },
          "tentative": false
        },
        {
          "title": "Characterising propaganda on Persian-language Telegram during the 2024 Iran-Israel war",
          "wid": "characterising-propaganda-on-persian-language-telegram-",
          "type": "publication",
          "year": 2026,
          "venue": "In preparation",
          "link": "",
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Telegram propaganda detection",
            "Persian-language MDH research",
            "Iran-Israel war 2024"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "Study of a self-collected dataset of Persian-language Telegram posts from the 2024 Iran-Israel war. It measures how much of the content is propaganda and which manipulation techniques are used, rather than judging whether individual claims are true, since there is no ground-truth fact-checking for very recent political events while propagandistic technique can still be characterised. It addresses a double gap: Telegram is increasingly used in Europe yet moderation-light, and Persian is under-represented in misinformation research.",
          "why": "It characterises wartime propaganda across an under-represented language and platform.",
          "data": "Text, ~3 million Persian-language Telegram posts (2024 Iran-Israel war)",
          "method_qualifiers": [
            "Data analysis"
          ]
        },
        {
          "title": "On-device cross-modal misinformation-detection prototype",
          "wid": "on-device-cross-modal-misinformation-detection-prototyp",
          "type": "project",
          "year": 2024,
          "venue": "In progress (two-year semester-project line)",
          "link": "",
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "Cross-modal misinformation",
            "On-device detection",
            "Local LLMs"
          ],
          "stage": "Monitoring",
          "relevance": 4,
          "about": "Prototype that flags misinformation directly on a user's phone or browser, using local language models so nothing leaves the device. It targets the hardest cross-modal cases, where text and an image are each individually true but their pairing is fabricated, for example a real bombing photo attached to a false claim about who carried out the strike. Single-modality checkers pass both pieces, so the deception sits in the combination, and the team reports that state-of-the-art tools do poorly on this data. It is currently a work in progress at roughly 60 percent accuracy.",
          "why": "It detects cross-modal misinformation where a true image and true text are deceptively paired.",
          "data": "Text and Image, social-media posts (Instagram / Twitter / Telegram / WhatsApp focus)",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Detector robustness",
            "On-device flagging",
            "Out-of-context image detection"
          ],
          "key_terms": [
            "Cross-modal misinformation",
            "On-device inference",
            "Local language models",
            "On-device detection",
            "Image-text mismatch"
          ],
          "models": [],
          "method_qualifiers": [
            "On-device inference",
            "Cross-modal attention",
            "Local model deployment"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "Text and image pairs from social media posts.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Instagram",
            "Telegram",
            "Twitter",
            "WhatsApp"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Out-of-context image detection",
              "On-device flagging",
              "Detector robustness"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "Detector robustness",
              "Out-of-context image detection"
            ]
          },
          "tentative": true
        },
        {
          "title": "Fighting Disinformation & the Role of AI (UNIL HEC workshop, 24 September 2026)",
          "wid": "24-september-2026-unil-workshop-misinformation-in-the-a",
          "type": "event",
          "year": 2026,
          "venue": "HEC Lausanne, Research Center for Grand Challenges + Department of Information Systems",
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "UNIL event",
            "AI misinformation",
            "Cross-actor workshop"
          ],
          "stage": "NA",
          "relevance": 4,
          "about": "Full-day interdisciplinary workshop on disinformation and the role of AI, scheduled for 24 September 2026 at UNIL. It uses a keynote-plus-young-researcher-abstracts format and invites practitioners to brainstorm alongside researchers. As a cross-organised event, it brings the small Swiss MDH community together.",
          "why": "A public event dedicated to disinformation and the role of AI.",
          "data": "",
          "themes": [
            "Media literacy & public resilience"
          ],
          "subtopics": [],
          "key_terms": [
            "Disinformation",
            "Swiss MDH community",
            "UNIL HEC",
            "AI",
            "Workshop"
          ],
          "models": [],
          "method_qualifiers": [],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "No empirical data (event description).",
          "region_country": [
            "Switzerland"
          ],
          "label_version": "v5",
          "theme_qualifiers": {},
          "theme_qualifiers_canonical": {},
          "tentative": true
        },
        {
          "title": "Swiss Association of Journalists training on AI-for-journalism",
          "wid": "swiss-association-of-journalists-training-on-ai-for-jou",
          "type": "event",
          "year": "",
          "venue": "Swiss Association of Journalists",
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "Journalism training",
            "AI tools",
            "Swiss media"
          ],
          "stage": "Prevention",
          "relevance": 2,
          "about": "Training for journalists on using AI tools in newsrooms, cited as a model for the kind of journalist-facing education EPFL and UNIL should keep doing. It sits inside a wider Swiss AI-for-journalism training ecosystem, where the Center for Digital Trust acts as a matchmaker between EPFL labs and media outlets.",
          "why": "It builds newsroom capacity to handle AI-generated content.",
          "data": "",
          "themes": [
            "Media literacy & public resilience"
          ],
          "subtopics": [
            "Practitioner preparedness"
          ],
          "key_terms": [
            "AI-for-journalism",
            "Journalist training",
            "Swiss media",
            "Newsroom capacity",
            "AI tools in newsrooms"
          ],
          "models": [],
          "method_qualifiers": [],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "No empirical data (journalist training event).",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "Media literacy & public resilience": [
              "Practitioner preparedness"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media literacy & public resilience": [
              "Practitioner preparedness"
            ]
          },
          "tentative": true
        },
        {
          "title": "The Effect of Education in Prompt Engineering: Evidence from Journalists",
          "wid": "the-effect-of-education-in-prompt-engineering-evidence-",
          "type": "publication",
          "year": 2026,
          "venue": "ICWSM 2026",
          "link": "https://doi.org/10.1609/icwsm.v20i1.42634",
          "authors": [
            "Amirsiavosh Bashardoust",
            "Yuanjun Feng",
            "Dominique Geissler",
            "Stefan Feuerriegel",
            "Yash Raj Shrestha"
          ],
          "epfl_authors": [
            "Amirsiavosh Bashardoust",
            "Yuanjun Feng",
            "Yash Raj Shrestha"
          ],
          "about": "Experiment testing whether training in prompt engineering improves how journalists interact with LLMs. In the study, 29 science journalists used ChatGPT-3.5 to write short Twitter/X posts about scientific articles before and after a 2-hour in-person training, and the posts were then rated for accuracy by domain experts and for text quality by 285 non-expert readers. Training significantly improved the journalists' perceived expertise in using LLMs, while the perceived helpfulness of LLMs tended to decline, though not significantly. Expert ratings moved in opposite directions for the two articles and reader ratings varied across text quality dimensions; the authors call their findings mixed, with some post-hoc analyses lacking statistically significant results.",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Factual accuracy",
            "Hallucination",
            "Scientific misrepresentation"
          ],
          "mdh_focus": [
            "M"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "Prevention",
          "key_terms": [
            "Prompt engineering training",
            "AI literacy",
            "Science communication",
            "Journalism",
            "Hallucination"
          ],
          "models": [
            "ChatGPT-3.5"
          ],
          "method_qualifiers": [
            "Mixed-effects regression",
            "Within-subject experiment",
            "Chain-of-thought",
            "Few-shot prompting"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "29 journalists, 285 readers, social media posts, survey responses.",
          "platform": [
            "Twitter/X"
          ],
          "region_country": [
            "Switzerland"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Hallucination",
              "Factual accuracy",
              "Scientific misrepresentation"
            ]
          },
          "theme_qualifiers_canonical": {},
          "tentative": false
        }
      ]
    },
    {
      "id": "zavolokina",
      "name": "Liudmila Zavolokina",
      "unit": "Digital Innovation Lab, UNIL HEC",
      "faculty": "External (UNIL HEC)",
      "mdh_focus": [
        "D",
        "M"
      ],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "HCI",
        "Design Science Research",
        "Propaganda Detection",
        "LLMs",
        "Media Literacy Tools",
        "Critical-Thinking Support",
        "User Studies"
      ],
      "stage": "Prevention + Monitoring",
      "publications": [
        {
          "title": "Biased by Design, Leveraging Inherent AI Biases to Enhance Critical Thinking of News Readers",
          "wid": "biased-by-design-leveraging-ai-biases-to-enhance-critic",
          "type": "publication",
          "year": 2025,
          "link": "https://aisel.aisnet.org/ecis2025/hci/hci/8/",
          "authors": [
            "Liudmila Zavolokina (UNIL)",
            "Kilian Sprenkamp (UZH)",
            "Zoya Katashinskaya (UZH)",
            "Daniel Gordon Jones (UZH)"
          ],
          "epfl_authors": [
            "Liudmila Zavolokina (UNIL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "AI bias awareness",
            "Critical-thinking tools"
          ],
          "stage": "Prevention + Monitoring",
          "relevance": 5,
          "about": "Exploratory HCI study of whether deliberately surfacing the political biases of an LLM-based news tool can prompt readers to engage more critically with its AI-generated explanations. In a user study, participants generally did not notice biases in the explanations until the researchers raised them in interview, after which participants engaged more actively in the discussion. It sets out three design recommendations: make bias explicit, let users control personalization based on their own political perspective, and introduce diverse viewpoints gradually.",
          "why": "It studies an AI news tool meant to build reader critical thinking against biased or slanted news framing, and how readers perceive the tool's own political bias.",
          "data": "Text",
          "themes": [
            "Media literacy & public resilience",
            "Persuasion & cognitive effects",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Cognitive dissonance",
            "Confirmation bias",
            "Critical thinking support",
            "Motivated reasoning",
            "Trust indicators"
          ],
          "key_terms": [
            "Propaganda detection",
            "Critical thinking",
            "Browser extension",
            "AI bias",
            "Cognitive dissonance"
          ],
          "models": [
            "GPT-4"
          ],
          "method_qualifiers": [
            "Thematic analysis"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "20 semi-structured interviews with German news readers, 594 minutes.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "Germany"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Media literacy & public resilience": [
              "Critical thinking support",
              "Credibility and bias indicators"
            ],
            "Persuasion & cognitive effects": [
              "Motivated reasoning",
              "Cognitive dissonance",
              "Confirmation bias"
            ],
            "Verification & content authenticity": [
              "Trust indicators"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media literacy & public resilience": [
              "Credibility and bias indicators",
              "Critical thinking support"
            ],
            "Persuasion & cognitive effects": [
              "Motivated reasoning"
            ],
            "Verification & content authenticity": [
              "Bias and framing analysis",
              "Trust indicators"
            ]
          },
          "tentative": false
        },
        {
          "title": "Effective Yet Ephemeral Propaganda Defense, There Needs to Be More than One Shot Inoculation to Enhance Critical Thinking",
          "wid": "effective-yet-ephemeral-propaganda-defense-there-needs-",
          "type": "publication",
          "year": 2025,
          "link": "https://arxiv.org/abs/2503.16497",
          "authors": [
            "Nicolas Hoferer (UZH)",
            "Kilian Sprenkamp (UZH)",
            "Dorian Christoph Quelle (UZH)",
            "Daniel Gordon Jones (UZH)",
            "Zoya Katashinskaya (UZH)",
            "Alexandre Bovet (UZH)",
            "Liudmila Zavolokina (UNIL)"
          ],
          "epfl_authors": [
            "Liudmila Zavolokina (UNIL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Propaganda detection",
            "Media literacy",
            "Tool dependency"
          ],
          "stage": "Prevention + Monitoring",
          "relevance": 5,
          "about": "Study testing whether an LLM-based propaganda-detection reading tool produces critical thinking that lasts after use. A two-phase online experiment combining inoculation theory with a dual-system model measured critical thinking during use and again one week later. The tool raised self-reported critical thinking while in use and slowed reading by about 50 seconds per article, but one week later showed no statistically significant difference from the control group. The pattern indicates a single exposure does not build durable resistance to propaganda.",
          "why": "It evaluates how well a propaganda-defence tool builds lasting reader resistance to misinformation.",
          "data": "Text",
          "themes": [
            "Media literacy & public resilience",
            "Persuasion & cognitive effects",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Attitude change",
            "Bias and framing analysis",
            "Dual-process thinking",
            "Psychological inoculation",
            "Skill retention"
          ],
          "key_terms": [
            "Propaganda detection",
            "Inoculation theory",
            "Critical thinking",
            "Browser extension",
            "Apollolytics"
          ],
          "models": [
            "GPT-4o"
          ],
          "method_qualifiers": [
            "Randomized controlled trial",
            "ReAct prompting",
            "Causal inference",
            "Between-subjects experiment"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "follow_up": [
            {
              "what": "Apollolytics, the propaganda detection and contextualization browser extension studied in this paper, released publicly on the Chrome Web Store and Firefox Add-ons (compatible with Edge, Brave and Opera)",
              "kind": "deployment",
              "url": "https://apollolytics.com/",
              "evidence": "apollolytics.com lists the three papers behind the tool, the third being \"Effective Yet Ephemeral Propaganda Defense\", described on the site as \"Investigating why one-shot inoculation is insufficient for long-term propaganda recognition\", and states the tool is \"developed by researchers of the University of Zurich and the University of Lausanne\". The DIZH project closure page adds: \"As part of the project, the browser extension Apollolytics was developed to help users recognize and critically evaluate propagandistic content in online news.\" and \"The research results were presented at CHI 2025 and the full article was published at ECIS 2025.\" The Firefox listing (https://addons.mozilla.org/en-US/firefox/addon/apollolytics/, developer Daniel Gordon Jones, a co-author) reads: \"Apollolytics is an extension that enables users to check text on websites for the presence of propaganda techniques and to contextualise claims within.\""
            },
            {
              "what": "Biased by Design: Leveraging AI Biases to Enhance Critical Thinking of News Readers (Zavolokina, Sprenkamp, Katashinskaya, Jones), Completed Research Paper, 33rd European Conference on Information Systems (ECIS 2025), Amman, Jordan - carries the same tool forward by turning the LLM's political bias into a critical-thinking device via model choice and stance personalization",
              "kind": "successor-work",
              "url": "https://api.unil.ch/iris/server/api/core/bitstreams/0dbcc680-ee75-4d4b-b7b5-ece1bd0096ec/content",
              "evidence": "ECIS 2025 paper, Section 2 Case description: \"This work builds upon prior research on designing a propaganda detection tool named Apollolytics that nudges its users, i.e., readers of digital news, to think more critically about the news content they consume while reading it (Hoferer et al., 2025; Sprenkamp et al., 2023; Zavolokina et al., 2024).\" Its reference list resolves Hoferer et al., 2025 as \"Hoferer, N., Sprenkamp, K., Quelle, D. C., Jones, D. G., Katashinskaya, Z., Bovet, A., & Zavolokina, L. (2025). Effective Yet Ephemeral Propaganda Defense: There Needs to Be More than One-Shot Inoculation to Enhance Critical Thinking. Extended Abstracts of the CHI Conference on Human Factors in Computing Systems (CHI EA '25) ... https://doi.org/10.1145/3706599.3720125\", i.e. this work."
            }
          ],
          "data_description": "500+ Prolific participants, 3 news articles, 2 surveys over 1 week.",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "Media literacy & public resilience": [
              "Critical thinking support",
              "Psychological inoculation",
              "Skill retention"
            ],
            "Persuasion & cognitive effects": [
              "Dual-process thinking",
              "Attitude change"
            ],
            "Verification & content authenticity": [
              "Bias and framing analysis"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media literacy & public resilience": [
              "Critical thinking support",
              "Psychological inoculation",
              "Skill retention"
            ],
            "Persuasion & cognitive effects": [
              "Attitude change",
              "Dual-process thinking",
              "Trust in AI tools"
            ],
            "Verification & content authenticity": [
              "Bias and framing analysis",
              "Claim credibility inference",
              "Evidence contextualisation"
            ]
          },
          "tentative": false
        },
        {
          "title": "Think Fast, Think Slow, Think Critical, Designing an Automated Propaganda Detection Tool (ClarifAI)",
          "wid": "think-fast-think-slow-think-critical-designing-an-autom",
          "type": "publication",
          "year": 2024,
          "link": "https://arxiv.org/abs/2402.19135",
          "authors": [
            "Liudmila Zavolokina",
            "Kilian Sprenkamp",
            "Zoya Katashinskaya",
            "Daniel Gordon Jones",
            "Gerhard Schwabe"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Propaganda detection",
            "Critical thinking",
            "HCI"
          ],
          "stage": "Prevention + Monitoring",
          "relevance": 5,
          "about": "Design study of a browser-based tool, ClarifAI, that uses a large language model to flag propagandistic statements in news articles and show short explanations when a reader hovers over a flagged passage. Grounded in dual-system thinking, it compares a no-tool baseline, a highlight-only version, and a full highlight-and-explain version. Explanations, not just highlighting, drove more critical reading: full-version reading time was 199.4 seconds per article versus 150.7 for the basic version, self-reported propaganda awareness rose from 82 to 96 percent, detection cost about 0.16 US dollars per article, and expert raters agreed with the tool on 65 percent of cases.",
          "why": "It detects propagandistic statements and aims to improve how critically people read news.",
          "data": "Text",
          "themes": [
            "Media literacy & public resilience",
            "Persuasion & cognitive effects",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Digital nudging",
            "Dual-process thinking",
            "Trust indicators"
          ],
          "key_terms": [
            "Propaganda detection",
            "Digital nudging",
            "Dual-system thinking",
            "Russia-Ukraine war coverage",
            "Critical thinking"
          ],
          "models": [
            "GPT-4"
          ],
          "method_qualifiers": [
            "Few-shot prompting",
            "Design science research",
            "Between-subjects experiment",
            "Randomized controlled trial"
          ],
          "events_cases": [
            "Russian aggression in Ukraine"
          ],
          "built_at_epfl": [
            {
              "name": "ClarifAI",
              "kind": "tool",
              "url": "https://apollolytics.com",
              "evidence": "apollolytics.com, \"Our Research\" section, lists this paper as one of three and describes it as: \"Think Fast, Think Slow, Think Critical - Designing the first iteration of Apollolytics and demonstrating its ability to enhance critical thinking.\" The paper itself flags the name as provisional: footnote 1 reads \"The name is tentatively used for this research.\""
            }
          ],
          "follow_up": [
            {
              "what": "Apollolytics, the shipped propaganda-detection browser extension that ClarifAI became, published on the Chrome Web Store and Firefox Add-ons",
              "kind": "deployment",
              "url": "https://apollolytics.com",
              "evidence": "apollolytics.com: \"Apollolytics automatically detects propaganda techniques, empowering you to think critically about the media you consume.\" and \"Apollolytics is developed by researchers of the University of Zurich and the University of Lausanne.\" Its research section names this paper as \"Designing the first iteration of Apollolytics\". Firefox listing (addons.mozilla.org/en-US/firefox/addon/apollolytics/, developer Daniel Gordon Jones, a co-author): \"Apollolytics is an extension that enables users to check text on websites for the presence of propaganda techniques and to contextualise claims within.\""
            },
            {
              "what": "DIZH-funded project \"Apollolytics: Automated Propaganda Detection in the Browser\" (formerly announced as \"ClarifAI: Making the Unseen Seen with AI-assisted Propaganda Detection and Fact-Checking\"), led by Liudmila Zavolokina, CHF 74945, from March 2024, 12 months",
              "kind": "project",
              "url": "https://www.dizh.uzh.ch/en/2025/09/25/project-closure-apollolytics-automated-propaganda-detection-in-the-browser/",
              "evidence": "DIZH project-closure page: the project produced a browser extension, Apollolytics, that uses large language models to identify propaganda techniques in online news; \"Studies have shown that using Apollolytics strengthens critical thinking and improves the ability to recognize propaganda.\" and \"The project was funded in the 1st Founder Call.\" The project announcement page (dizh.uzh.ch, 21 Dec 2023) gives lead Dr. Liudmila Zavolokina (UZH Digital Society Initiative), CHF 74945, March 2024 + 12 months, with partners including Prof. Alexandre Bovet and Kilian Sprenkamp."
            }
          ],
          "data_description": "Expert review of 107 flagged passages; online experiment with 248 participants.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "Russia",
            "Ukraine",
            "United Kingdom",
            "United States"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Media literacy & public resilience": [
              "Critical thinking support",
              "Credibility and bias indicators"
            ],
            "Persuasion & cognitive effects": [
              "Digital nudging",
              "Dual-process thinking",
              "Trust in AI tools"
            ],
            "Verification & content authenticity": [
              "Trust indicators"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media literacy & public resilience": [
              "Credibility and bias indicators",
              "Critical thinking support"
            ],
            "Persuasion & cognitive effects": [
              "Digital nudging",
              "Dual-process thinking",
              "Trust in AI tools"
            ],
            "Verification & content authenticity": [
              "Bias and framing analysis",
              "Trust indicators"
            ]
          },
          "tentative": false
        },
        {
          "title": "Large Language Models for Propaganda Detection",
          "wid": "large-language-models-for-propaganda-detection",
          "type": "publication",
          "year": 2023,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2310.06422",
          "authors": [
            "Kilian Sprenkamp",
            "Daniel Gordon Jones",
            "Liudmila Zavolokina"
          ],
          "about": "Written at the University of Zurich, before Liudmila Zavolokina joined the University of Lausanne in 2024. Study testing large language models for propaganda detection, framed as multi-label classification of news articles from the SemEval-2020 task 11 dataset labeled with 14 propaganda techniques. Five variations are compared with a RoBERTa baseline: GPT-4 used out of the box with a 'base' or a 'chain of thought' prompt, and fine-tuned GPT-3 with those prompts or without instruction. GPT-4 'base' reaches the highest F1 score among them, 58.11 percent, below the baseline's 63.40 percent but, in a comparison the authors call limited, higher on seven of 14 techniques. Fine-tuned GPT-3 models tended to overfit to the training data and showed repetitions, hallucinations and unanticipated terminations, while the chain of thought prompt raised GPT-4's precision.",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Bias and framing analysis",
            "Propaganda technique detection",
            "Trust indicators"
          ],
          "mdh_focus": [
            "D"
          ],
          "mdh_relevance": "direct",
          "relevance": 5,
          "stage": "Monitoring",
          "key_terms": [
            "Propaganda detection",
            "Propaganda techniques",
            "Multi-label classification",
            "News articles",
            "Prompt engineering"
          ],
          "models": [
            "GPT-3",
            "GPT-3 Davinci",
            "GPT-4",
            "RoBERTa"
          ],
          "method_qualifiers": [
            "Few-shot prompting",
            "Fine-tuning"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "371 training, 75 testing news articles from SemEval-2020 task 11.",
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Bias and framing analysis",
              "Trust indicators",
              "Propaganda technique detection"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "Bias and framing analysis",
              "Trust indicators"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "holstein",
      "name": "Ken Holstein",
      "url": "https://people.epfl.ch/ken.holstein",
      "unit": "CoALA Lab",
      "faculty": "IC",
      "mdh_focus": [
        "M",
        "D",
        "H"
      ],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "HCI",
        "ML Auditing",
        "Algorithmic Fairness",
        "Human-AI Collaboration"
      ],
      "stage": "Monitoring + Mitigation",
      "publications": [
        {
          "title": "PersonaTeaming: Exploring How Introducing Personas Can Improve Automated AI Red-Teaming",
          "wid": "personateaming-exploring-how-introducing-personas-can-i",
          "type": "publication",
          "year": 2025,
          "venue": "NeurIPS 2025 Workshop on Regulatable ML",
          "link": "https://arxiv.org/abs/2509.03728",
          "authors": [
            "Wesley Hanwen Deng",
            "Sunnie S. Y. Kim",
            "Akshita Jha",
            "Ken Holstein",
            "Motahhare Eslami",
            "Lauren Wilcox",
            "Leon A Gatys"
          ],
          "about": "Ken Holstein's affiliation on this paper is Carnegie Mellon University, before his appointment at EPFL. Paper presenting PersonaTeaming, an automated AI red-teaming method that introduces personas, either red-teaming experts or regular AI users, into adversarial prompt mutation, along with an algorithm that generates personas adapted to each seed prompt and new metrics for mutation distance. Built on the RainbowPlus method and run with GPT-4o on HarmBench seed prompts, every condition adding personas to RainbowPlus raises attack success rates, by up to 144.1 percent, and most maintain or improve diversity scores. Red-teamer personas usually reach higher attack success than user personas, which produce more diverse prompts, and dynamic persona generation achieves higher diversity than fixed personas while maintaining high attack success.",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Adaptive attacks",
            "Automated red-teaming",
            "Misuse risk assessment"
          ],
          "mdh_focus": [
            "D",
            "H"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "Prevention",
          "key_terms": [
            "Automated red-teaming",
            "Adversarial prompt generation",
            "Adversarial prompt mutation",
            "AI safety",
            "Attack success rate"
          ],
          "models": [
            "GPT-4o",
            "all-MiniLM-L6-v2"
          ],
          "method_qualifiers": [
            "Jailbreaking",
            "Adversarial prompting",
            "LLM-as-a-judge"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "PersonaTeaming",
              "kind": "tool"
            }
          ],
          "data_description": "9 conditions x 2000 mutated prompts each, from up to 150 HarmBench seed prompts.",
          "targeted_group": [
            "Racial or ethnic minorities",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Adaptive attacks",
              "Misuse risk assessment",
              "Automated red-teaming"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Adaptive attacks",
              "Misuse risk assessment"
            ]
          },
          "tentative": false
        },
        {
          "title": "Policy Maps: Tools for Guiding the Unbounded Space of LLM Behaviors",
          "wid": "policy-maps-tools-for-guiding-the-unbounded-space-of-ll",
          "type": "publication",
          "year": 2025,
          "venue": "ACM UIST 2025",
          "link": "https://arxiv.org/abs/2409.18203",
          "authors": [
            "Michelle S. Lam",
            "Fred Hohman",
            "Dominik Moritz",
            "Jeffrey P. Bigham",
            "Kenneth Holstein",
            "Mary Beth Kery"
          ],
          "about": "Ken Holstein's affiliation on this paper is Carnegie Mellon University, before his appointment at EPFL. Paper introducing policy maps, an approach to designing policy for large language models inspired by physical mapmaking, and Policy Projector, an interactive tool for surveying model input-output pairs, defining custom regions such as violence, and navigating them with if-then policy rules that act on LLM outputs. In an evaluation with 12 AI safety experts, participants crafted policies around problematic model behaviors such as incorrect gender assumptions, authoring 24 new policies that drew on 43 concepts. A technical evaluation found that the implementation matches cases to concepts with an accuracy of 85.8 percent, and that model steering significantly reduces positive concept classifications by an LLM classifier.",
          "themes": [
            "AI safety",
            "Content moderation & enforcement"
          ],
          "subtopics": [
            "If-then moderation rules",
            "LLM concept classification",
            "LLM policy authoring",
            "Misuse risk assessment",
            "Safety alignment"
          ],
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 3,
          "stage": "Prevention",
          "key_terms": [
            "LLM safety policy",
            "Policy gap identification",
            "Model steering",
            "AI governance",
            "AI policy"
          ],
          "models": [
            "GPT-4o",
            "GPT-4o-mini",
            "Meta-Llama-3-8B-Instruct",
            "all-MiniLM-L6-v2",
            "gpt-4o",
            "gpt-4o-mini",
            "text-embedding-3-large"
          ],
          "method_qualifiers": [
            "Few-shot prompting",
            "Representation finetuning",
            "Concept induction"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Policy maps",
              "kind": "framework"
            },
            {
              "name": "Policy Projector",
              "kind": "tool"
            }
          ],
          "data_description": "400 red-team input-output pairs, 12 AI safety experts, policy rules.",
          "region_country": [
            "European Union",
            "United States"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Safety alignment",
              "LLM policy authoring",
              "Misuse risk assessment"
            ],
            "Content moderation & enforcement": [
              "If-then moderation rules",
              "Policy rule authoring",
              "LLM concept classification"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Misuse risk assessment",
              "Safety alignment"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "pearl_pu",
      "name": "Pearl Pu",
      "url": "https://people.epfl.ch/pearl.pu",
      "unit": "HCI Lab",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "Social Computing",
        "HCI",
        "ML"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "schrimpf",
      "name": "Martin Schrimpf",
      "url": "https://people.epfl.ch/martin.schrimpf",
      "unit": "NeuroAI Lab",
      "faculty": "IC / SV (Neuro-X)",
      "mdh_focus": [],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "Neuroscience-inspired AI",
        "Language Models",
        "ML"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "hartley",
      "name": "Mary-Anne Hartley",
      "url": "https://people.epfl.ch/mary-anne.hartley",
      "unit": "LiGHT",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "ML",
        "Healthcare AI",
        "NLP"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "kamgarpour",
      "name": "Maryam Kamgarpour",
      "url": "https://people.epfl.ch/maryam.kamgarpour",
      "unit": "SYCAMORE",
      "faculty": "STI",
      "mdh_focus": [],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "Control",
        "Optimisation",
        "Game Theory",
        "Multi-agent Systems"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "abbe",
      "name": "Emmanuel Abbe",
      "url": "https://people.epfl.ch/emmanuel.abbe",
      "unit": "MDS",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [
        "Data-agnostic"
      ],
      "techTypes": [
        "ML Theory"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "chiesa",
      "name": "Alessandro Chiesa",
      "url": "https://people.epfl.ch/alessandro.chiesa",
      "unit": "COMPSEC",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [],
      "techTypes": [
        "Cryptography",
        "Zero-Knowledge Proofs",
        "Content Provenance",
        "Credential Authentication"
      ],
      "stage": "Prevention",
      "publications": []
    },
    {
      "id": "gulcehre",
      "name": "Caglar Gulcehre",
      "url": "https://people.epfl.ch/caglar.gulcehre",
      "unit": "CLAIRE",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "Deep Reinforcement Learning",
        "LLMs",
        "AI Alignment"
      ],
      "stage": "Prevention",
      "publications": []
    },
    {
      "id": "raynal",
      "name": "Mathilde Raynal",
      "url": "https://people.epfl.ch/mathilde.raynal",
      "unit": "SPRING Lab",
      "faculty": "IC",
      "mdh_focus": [
        "H"
      ],
      "dataTypes": [
        "Text",
        "Image"
      ],
      "techTypes": [
        "Content Moderation",
        "Adversarial ML",
        "Toxicity Detection",
        "Privacy Engineering",
        "Dataset Construction",
        "HCI"
      ],
      "stage": "Prevention",
      "publications": [
        {
          "title": "Adversarial evasion of deployed content-moderation classifiers",
          "wid": "adversarial-evasion-of-deployed-content-moderation-clas",
          "type": "publication",
          "year": "Ongoing",
          "venue": "SPRING Lab, EPFL (in progress, thesis target 2027)",
          "epfl_authors": [
            "Mathilde Raynal (SPRING, EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "H"
          ],
          "stage": "Monitoring",
          "relevance": 4,
          "about": "Ongoing work showing that motivated, low-resource adversaries can evade deployed content-moderation classifiers (Perspective API and Hugging Face models) while preserving the semantic meaning of the harmful content. Unpublished; thesis target early 2027.",
          "mdh_topics": [
            "Content moderation",
            "Adversarial evasion",
            "Toxicity detection",
            "Perspective API"
          ]
        },
        {
          "title": "Context-aware content moderation (Reddit-with-context dataset)",
          "wid": "context-aware-content-moderation-reddit-with-context-da",
          "type": "publication",
          "year": "Ongoing",
          "venue": "SPRING Lab, EPFL (in progress)",
          "epfl_authors": [
            "Mathilde Raynal (SPRING, EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "H"
          ],
          "stage": "Mitigation",
          "relevance": 4,
          "about": "Ongoing work on the context-aware-moderation gap: how content moderation fails when it ignores conversational context. Building a Reddit-with-context dataset to study this. Unpublished.",
          "mdh_topics": [
            "Content moderation",
            "Context-aware moderation",
            "Reddit",
            "Dataset construction"
          ]
        }
      ]
    },
    {
      "id": "kuncak",
      "name": "Viktor Kunčak",
      "url": "https://people.epfl.ch/viktor.kuncak",
      "unit": "LARA",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [],
      "techTypes": [
        "Formal Verification",
        "Programming Languages",
        "Automated Reasoning"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "zdeborova",
      "name": "Lenka Zdeborová",
      "url": "https://people.epfl.ch/lenka.zdeborova",
      "unit": "SPOC",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [
        "Data-agnostic"
      ],
      "techTypes": [
        "Statistical Physics",
        "ML Theory"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "brbic",
      "name": "Maria Brbić",
      "url": "https://people.epfl.ch/maria.brbic",
      "unit": "MLBio",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [],
      "techTypes": [
        "ML",
        "Computational Biology",
        "Representation Learning"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "ailamaki",
      "name": "Anastasia Ailamaki",
      "url": "https://people.epfl.ch/anastasia.ailamaki",
      "unit": "DIAS",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [],
      "techTypes": [
        "Data Systems",
        "Databases"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "candea",
      "name": "George Candea",
      "url": "https://people.epfl.ch/george.candea",
      "unit": "DSLAB",
      "faculty": "IC",
      "mdh_focus": [],
      "dataTypes": [],
      "techTypes": [
        "Dependable Systems",
        "Software Security"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "schonenberger",
      "name": "Klaus Schönenberger",
      "unit": "EssentialTech Centre",
      "faculty": "ENAC",
      "mdh_focus": [
        "M",
        "D",
        "H"
      ],
      "dataTypes": [
        "Data-agnostic"
      ],
      "techTypes": [
        "PeaceTech",
        "Humanitarian Innovation"
      ],
      "stage": "Prevention + Mitigation",
      "publications": [
        {
          "title": "PeaceTech Hackathon 2026",
          "wid": "peacetech-hackathon-2026",
          "type": "event",
          "year": 2026,
          "venue": "EPFL Campus, Lausanne, 26-27 September 2026",
          "link": "https://peacetech.ch/",
          "authors": [
            "Stefania Pia Grottola"
          ],
          "epfl_authors": [
            "Stefania Pia Grottola"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_topics": [
            "PeaceTech",
            "Hackathon",
            "Gender-based violence",
            "Social cohesion",
            "Humanitarian innovation"
          ],
          "stage": "NA",
          "relevance": 5,
          "about": "Second edition of an EPFL hackathon run by the EssentialTech Centre and its PeaceTech Division. The 2026 edition is themed on misinformation, disinformation and hate speech, with an emphasis on gender-based violence. It convenes participants from academia, technology, NGOs and international organisations to work on challenges from partner organisations and design tools that support peace and social cohesion.",
          "why": "The 2026 edition is themed on misinformation, disinformation and hate speech with a gender-based-violence emphasis.",
          "data": "not applicable (hackathon)",
          "lab": "EssentialTech Centre",
          "themes": [
            "Gender-based violence & misogyny"
          ],
          "subtopics": [
            "Gendered victimisation",
            "Technology-facilitated abuse"
          ],
          "key_terms": [
            "Gendered disinformation",
            "Technology-facilitated gender-based violence",
            "PeaceTech",
            "PeaceTech Hackathon",
            "Technology-facilitated GBV"
          ],
          "models": [],
          "method_qualifiers": [],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "No empirical data (event description).",
          "follow_up_checked": "2026-09-02",
          "targeted_group": [
            "Children",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Gender-based violence & misogyny": [
              "Gendered victimisation",
              "Technology-facilitated abuse"
            ]
          },
          "theme_qualifiers_canonical": {
            "Gender-based violence & misogyny": [
              "Gendered victimisation"
            ]
          },
          "tentative": true
        },
        {
          "title": "Science and technology: a framework for peace",
          "wid": "science-and-technology-a-framework-for-peace",
          "type": "publication",
          "year": 2024,
          "link": "https://doi.org/10.1038/s44172-024-00310-4",
          "authors": [
            "Mariazel Maqueda López",
            "Sheena Kennedy",
            "Solomzi Makohliso",
            "Yves Daccord",
            "Klaus Schönenberger"
          ],
          "epfl_authors": [
            "Mariazel Maqueda López",
            "Sheena Kennedy",
            "Solomzi Makohliso",
            "Yves Daccord",
            "Klaus Schönenberger"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_topics": [
            "PeaceTech framework",
            "Cultural violence",
            "Do-no-harm"
          ],
          "stage": "Mitigation",
          "relevance": 2,
          "about": "Conceptual framework for how scientists and engineers can connect their work to peacebuilding, an area it calls PeaceTech. It rests on four principles: a common language drawn from Galtung's typology of direct, structural and cultural violence; needs-driven innovation rooted in affected populations; multi-stakeholder engagement across government, civil society, academia and industry; and the humanitarian do-no-harm principle. It places misinformation, disinformation, hate speech and online harassment within the cultural-violence layer, noting how digital platforms can spread content that deepens divides and opens people to manipulation. It cites a SIPRI press release for the figure that global military expenditure reached 2.24 trillion US dollars in 2023, with under 1 percent going to peacebuilding. All five authors are at the EPFL EssentialTech Centre.",
          "why": "It places countering disinformation and hate speech within its cultural-violence layer and anchors the MDH mapping project.",
          "data": "Conceptual / framework",
          "themes": [
            "Platform governance & regulation"
          ],
          "subtopics": [],
          "key_terms": [
            "PeaceTech",
            "Cultural violence",
            "Dual-use technology",
            "Conflict prevention",
            "cultural violence"
          ],
          "models": [],
          "method_qualifiers": [],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "No empirical data (policy/framework analysis).",
          "targeted_group": [
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {},
          "theme_qualifiers_canonical": {},
          "tentative": false
        },
        {
          "title": "Digital Dilemmas: Humanitarian Consequences (exhibition)",
          "wid": "digital-dilemmas-humanitarian-consequences-exhibitio",
          "type": "event",
          "year": 2024,
          "venue": "EPFL Pavilions, 3 May to 14 July 2024. EPFL EssentialTech Centre with the ICRC, co-organised with EPFL Pavilions, in partnership with the Center for Digital Trust (C4DT)",
          "link": "https://epfl-pavilions.ch/fr/exhibitions/digital-dilemmas",
          "authors": [
            "Grégoire Castella (EssentialTech)",
            "Louis Potter (EssentialTech)",
            "Klaus Schönenberger (EssentialTech)",
            "Sarah Kenderdine (EPFL Pavilions)",
            "Marie Carrard (EPFL Pavilions)",
            "Peter Grönquist (IVRL)",
            "Stéphanie Milliquet (C4DT)",
            "Philippe Stoll (ICRC)",
            "Catherina Zazzini (ICRC)",
            "Fabrice Lauper"
          ],
          "epfl_authors": [
            "Grégoire Castella (EssentialTech)",
            "Louis Potter (EssentialTech)",
            "Klaus Schönenberger (EssentialTech)",
            "Sarah Kenderdine (EPFL Pavilions)",
            "Marie Carrard (EPFL Pavilions)",
            "Peter Grönquist (IVRL)",
            "Stéphanie Milliquet (C4DT)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_topics": [
            "Public exhibition",
            "Digital ethics",
            "Awareness",
            "ICRC"
          ],
          "stage": "Prevention",
          "relevance": 3,
          "about": "Immersive public exhibition at EPFL Pavilions exploring the digital risks that civilian populations and humanitarian workers face in conflict zones, where digital tools open access to essential services but can also expose personal data, enable surveillance or worsen misinformation. Visitors worked through dilemmas covering biometrics, civilian involvement in digital warfare, misinformation, AI-generated deepfakes, algorithmic decision-making, connectivity and data protection, and is followed by an installation of solutions developed by EPFL laboratories with the ICRC and ETH Zurich. It first appeared at UN headquarters in New York before this expanded Swiss presentation.",
          "why": "It addresses misinformation and AI deepfakes in humanitarian settings, raising public awareness.",
          "data": "not applicable (public exhibition)",
          "lab": [
            "EssentialTech Centre",
            "IVRL",
            "EPFL Pavilions",
            "C4DT",
            "ICRC"
          ],
          "themes": [
            "Media literacy & public resilience"
          ],
          "subtopics": [
            "Digital literacy",
            "Immersive demonstration",
            "Public awareness campaigns"
          ],
          "key_terms": [
            "Humanitarian action",
            "Deepfakes",
            "Hate speech",
            "Conflict zones",
            "Armed conflict"
          ],
          "models": [],
          "method_qualifiers": [],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Digital Dilemmas: Humanitarian Consequences",
              "kind": "platform"
            }
          ],
          "follow_up": [
            {
              "what": "Digital Dilemmas 2.0 - the \"Deepfake and You\" exhibit, built by the ICRC with EPFL's Image and Visual Representations Lab and shown in the foyer of the UN General Assembly building in New York",
              "kind": "successor-work",
              "url": "https://blogs.icrc.org/intercross/2025/03/20/digital-dilemmas-2-0-deepfakes/",
              "evidence": "ICRC Intercross podcast page, published 20 March 2025, titled \"Digital Dilemmas 2.0: Deepfakes\": \"We tour a deepfake exhibit created by the ICRC and L'Ecole Polytechnique Federale de Lausanne (EPFL) at the United Nations.\" In the transcript, EPFL research software engineer Peter Gronquist says: \"So we're at the UN right now. And I'm about to give you a short tour of this deep fake exhibit that we've been setting up with ICRC and EPFL.\" The page adds: \"The ICRC is working with EPFL to explain and find solutions to deepfakes and other technologies disseminating harmful information among populations in conflict areas.\""
            }
          ],
          "data_description": "No empirical data (exhibition description).",
          "follow_up_checked": "2026-09-02",
          "targeted_group": [
            "Civilian populations",
            "Civilians in conflict zones",
            "Humanitarian aid organisations",
            "Humanitarian aid workers"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Media literacy & public resilience": [
              "Public awareness campaigns",
              "Immersive demonstration",
              "Digital literacy"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media literacy & public resilience": [
              "Digital literacy",
              "Public awareness campaigns"
            ]
          },
          "tentative": true
        }
      ]
    },
    {
      "id": "cevher",
      "name": "Volkan Cevher",
      "url": "https://people.epfl.ch/volkan.cevher",
      "unit": "LIONS",
      "faculty": "STI",
      "mdh_focus": [],
      "dataTypes": [
        "Text",
        "Image",
        "Data-agnostic"
      ],
      "techTypes": [
        "Adversarial ML",
        "Robust ML",
        "Optimisation"
      ],
      "stage": "Prevention + Mitigation",
      "publications": []
    },
    {
      "id": "haack",
      "name": "Patrick Haack",
      "unit": "Center for Grand Challenges",
      "faculty": "External (UNIL HEC)",
      "mdh_focus": [
        "M",
        "D"
      ],
      "dataTypes": [
        "Text"
      ],
      "techTypes": [
        "Organisational Behaviour",
        "Media Analysis",
        "Reputation"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "kaplan",
      "name": "Frédéric Kaplan",
      "url": "https://people.epfl.ch/frederic.kaplan",
      "unit": "DHLAB",
      "faculty": "IC",
      "mdh_focus": [
        "M",
        "D"
      ],
      "dataTypes": [
        "Text",
        "Data-agnostic"
      ],
      "techTypes": [
        "Digital Humanities",
        "NLP",
        "LLMs",
        "Knowledge Infrastructure",
        "Historical Text Analysis",
        "Wikipedia Edit Analysis"
      ],
      "stage": "Prevention + Monitoring",
      "publications": [
        {
          "title": "Narrative wars: Detecting the weaponisation of cultural heritage in the digital sphere",
          "wid": "cross-narrative-wars-project",
          "type": "project",
          "year": 2026,
          "venue": "CROSS 2026 (EPFL-UNIL Collaborative Research on Science and Society). Led by Hamest Tamrazyan (CDH, DHI-GE) and Emanuela Boros (CDH, DHI, DHLAB) at EPFL with Stephanie Prezioso and Hanna Perekhonda (FSSP, IEP) at UNIL. Emanuela Boros left EPFL in January 2026.",
          "link": "https://www.epfl.ch/schools/cdh/cross-2026/",
          "authors": [
            "Hamest Tamrazyan (EPFL DHI)",
            "Emanuela Boros (EPFL DHLab)",
            "Stéphanie Prezioso (UNIL IEP)",
            "Hanna Perekhonda (UNIL IEP)"
          ],
          "epfl_authors": [
            "Hamest Tamrazyan (EPFL DHI)",
            "Emanuela Boros (EPFL DHLab)",
            "Stéphanie Prezioso (UNIL IEP)",
            "Hanna Perekhonda (UNIL IEP)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Narrative manipulation",
            "Wikipedia",
            "Conflict zones",
            "Language weaponisation",
            "Ukraine",
            "Nagorno-Karabakh"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "Multi-year DHLAB project studying how language is weaponised in geopolitical conflict zones by tracking edits to Wikipedia across the Ukrainian, Armenian, Azerbaijani and English editions, with a focus on Ukraine and Nagorno-Karabakh. The work concentrates on subtle word-level shifts that re-anchor historical narratives over time rather than overt fake news, and the analysis is designed to stay neutral about which side is editing.",
          "why": "It studies covert narrative manipulation and incremental rewriting of contested history through coordinated Wikipedia edits in conflict zones.",
          "data": "Text (hundreds of thousands of Wikipedia edits in Ukrainian, Armenian, Azerbaijani, English language editions)",
          "themes": [
            "Media framing & narrative analysis"
          ],
          "subtopics": [
            "Cultural heritage narratives",
            "Issue framing",
            "Narrative reframing",
            "Wikipedia editing manipulation"
          ],
          "key_terms": [
            "Cultural heritage weaponisation",
            "Wikipedia",
            "Armenia",
            "Ukraine",
            "Conflict narratives"
          ],
          "models": [],
          "method_qualifiers": [
            "Thematic analysis",
            "Historical interpretation"
          ],
          "events_cases": [
            "Ukraine"
          ],
          "built_at_epfl": [],
          "data_description": "Wikipedia articles and edits concerning Armenia and Ukraine.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Wikipedia"
          ],
          "region_country": [
            "Armenia",
            "Ukraine"
          ],
          "targeted_group": [
            "Armenians",
            "Ukrainians"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Influence operations & coordinated manipulation": [
              "Narrative reframing",
              "Wikipedia editing manipulation"
            ],
            "Media framing & narrative analysis": [
              "Narrative shift over time",
              "Issue framing",
              "Cultural heritage narratives"
            ]
          },
          "theme_qualifiers_canonical": {
            "Influence operations & coordinated manipulation": [
              "Narrative reframing"
            ],
            "Media framing & narrative analysis": [
              "Issue framing",
              "Narrative shift over time"
            ]
          },
          "tentative": true
        },
        {
          "title": "Impresso project",
          "wid": "impresso-project",
          "type": "project",
          "year": "ongoing",
          "venue": "SNSF-funded research programme",
          "link": "https://impresso-project.ch/",
          "authors": [
            "Maud Ehrmann (EPFL DHLab)",
            "Simon Clematide (UZH)",
            "Raphaelle Ruppen Coutaz (UNIL)",
            "Marten During (Uni Luxembourg C2DH)",
            "Emanuela Boros (EPFL DHLab)",
            "Pauline Conti (EPFL DHLab)",
            "Marina Butyrskaya Moyer (EPFL DHLab)",
            "Arthur Michelet (UNIL)",
            "Martin Grandjean (EPFL DHLab)",
            "Caio Mello (Uni Luxembourg C2DH)",
            "Cao Vy (Uni Luxembourg C2DH)",
            "Daniele Guido (Uni Luxembourg C2DH)",
            "Estelle Bunout (Uni Luxembourg C2DH)",
            "Ferdaous Affan (Uni Luxembourg C2DH)",
            "Kirill Mitsurov (Uni Luxembourg C2DH)",
            "Roman Kalyakin (Uni Luxembourg C2DH)",
            "Andrianos Michail (UZH)",
            "Juri Opitz (UZH)",
            "Kaspar Beelen (SAS, University of London)"
          ],
          "epfl_authors": [
            "Maud Ehrmann (EPFL DHLab)",
            "Emanuela Boros (EPFL DHLab)",
            "Pauline Conti (EPFL DHLab)",
            "Marina Butyrskaya Moyer (EPFL DHLab)",
            "Raphaelle Ruppen Coutaz (UNIL)",
            "Arthur Michelet (UNIL)",
            "Martin Grandjean (EPFL DHLab)"
          ],
          "mdh_relevance": "infrastructure",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "Historical newspapers",
            "Knowledge infrastructure",
            "Sourced corpora"
          ],
          "stage": "NA",
          "relevance": 2,
          "about": "Interdisciplinary research programme applying machine learning to historical media, semantically enriching and connecting digitised newspapers and radio broadcasts so researchers can explore them across languages, countries and decades. It produces public web applications, datasets and models. On the press side it complements Time Machine as historical knowledge infrastructure, building a stable, sourced record of past media.",
          "why": "It builds a stable, sourced corpus of historical press and broadcast media, an indirect counterweight to a less anchored information environment.",
          "data": "Historical newspaper corpora (Swiss-French / Luxembourgish / multilingual press)",
          "themes": [
            "NA"
          ],
          "subtopics": [],
          "key_terms": [
            "Radio archives",
            "Digital humanities",
            "Historical media",
            "NLP",
            "Europe"
          ],
          "models": [],
          "method_qualifiers": [
            "Semantic enrichment"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Impresso",
              "kind": "platform",
              "url": "https://impresso-project.ch/app/",
              "evidence": "https://impresso-project.ch/the-app/ : \"The Impresso web app (https://impresso-project.ch/app/) offers a graphical user interface for exploration and the compilation of research datasets. It builds on the application developed during the first Impresso project. As part of the second project, the web app is revised to facilitate access to different types of historical media, such as audio recordings, and various types of text, e.g., typescripts, transcribed speech, or radio programming schedules.\" The first project is confirmed as Kaplan's at https://www.epfl.ch/labs/dhlab/projects/impresso-old/ : \"impresso - Media monitoring of the past. Mining 200 years of historical newspapers\", PI Frederic Kaplan (DHLAB, EPFL), September 2017 - August 2020, SNSF Sinergia."
            },
            {
              "name": "Impresso corpus",
              "kind": "dataset"
            },
            {
              "name": "Impresso Data Lab",
              "kind": "tool"
            },
            {
              "name": "Impresso web application",
              "kind": "tool"
            }
          ],
          "follow_up": [
            {
              "what": "Impresso II - 'Media Monitoring of the Past II. Beyond Borders: Connecting Historical Newspapers and Radio', the funded continuation (SNSF 213585 Sinergia + FNR 17498891 INTER, September 2023 to February 2027)",
              "kind": "project",
              "url": "https://www.epfl.ch/labs/dhlab/impresso-media-monitoring-of-the-past-ii-beyond-borders-connecting-historical-newspapers-and-radio/",
              "evidence": "\"funded by the Swiss National Science Foundation (SNSF 213585) and the Luxembourg National Research Fund (FNR 17498891) as part of the Sinergia / INTER funding programmes from September 2023 until February 2027.\" Applicant institutions: DHLAB at EPFL, University of Lausanne History Department, University of Zurich Institute of Computational Linguistics, University of Luxembourg C2DH. That it continues the first project is stated at https://impresso-project.ch/the-app/ : \"It builds on the application developed during the first Impresso project. As part of the second project, the web app is revised...\""
            },
            {
              "what": "HIPE (Identifying Historical People, Places and other Entities) - a series of evaluation campaigns / shared tasks on named entity recognition and linking in multilingual historical documents, spun out of the project",
              "kind": "benchmark",
              "url": "https://impresso-project.ch/hipe/",
              "evidence": "\"Initiated during the first Impresso project, HIPE (Identifying Historical People, Places and other Entities) is a series of evaluation campaigns, or shared tasks, on named entity recognition and linking in multilingual historical documents.\" and \"The HIPE-eval initiative will be continued during the second Impresso project and will gradually evolve to include further information extraction tasks on historical documents.\""
            },
            {
              "what": "Impresso data lab - API-based data access and annotation services with executable Jupyter notebooks, built in the second project",
              "kind": "deployment",
              "url": "https://impresso-project.ch/datalab",
              "evidence": "https://impresso-project.ch/the-app/ , section '2. Impresso data lab': \"The forthcoming Impresso data lab is an infrastructure for data access and annotation services via APIs, along with their integration in executable Jupyter notebooks. The data lab will provide researchers with examples based on experiments with data-driven analyses of the Impresso corpus. Annotation services offer access to semantic indexing models and the ability to relate external data to the corpus.\""
            },
            {
              "what": "Impresso model and dataset releases on HuggingFace: 27 models (historical NER, named entity linking, topic inference, OCR quality assessment, ad classification, multilingual BERT variants trained on historical media) and 10 datasets (including the HIPE NER and NEL sets)",
              "kind": "model",
              "url": "https://huggingface.co/impresso-project",
              "evidence": "Organisation description: \"an interdisciplinary research project using machine learning to transform how historical media are processed, enriched, explored, and studied across modalities\", developing a web application and datalab giving access to a multilingual corpus of historical newspapers and radio broadcasts; funded by the Swiss National Science Foundation."
            },
            {
              "what": "Impresso open-source codebase on GitHub (impresso-frontend, impresso-text-acquisition, impresso-pipelines, impresso-datalab-notebooks, CLEF-HIPE-2020, impresso-schemas and others)",
              "kind": "repository",
              "url": "https://github.com/impresso",
              "evidence": "Organisation description: \"Impresso - Media Monitoring of the Past ... an interdisciplinary research project that uses machine learning to pursue a paradigm shift in the processing, semantic enrichment, representation, exploration and study of historical media across modalities, temporal, linguistic, and national borders.\" The org states funding for two periods, 2017-2020 and 2023-2027, and that \"It develops the Impresso Web App and the upcoming Impresso Datalab.\""
            }
          ],
          "data_description": "Corpus of European newspaper and radio archives, many countries and languages.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "Europe"
          ],
          "label_version": "v5",
          "theme_qualifiers": {},
          "theme_qualifiers_canonical": {},
          "tentative": true
        }
      ]
    },
    {
      "id": "metille",
      "name": "Sylvain Métille",
      "unit": "Department of Business Law and Tax Law",
      "faculty": "External (UNIL Law / dhCenter)",
      "mdh_focus": [],
      "dataTypes": [],
      "techTypes": [
        "Data Protection Law",
        "Cybercrime Law",
        "GDPR",
        "Platform Accountability"
      ],
      "stage": "Prevention + Mitigation",
      "publications": []
    },
    {
      "id": "alahi",
      "name": "Alexandre Alahi",
      "url": "https://people.epfl.ch/alexandre.alahi",
      "unit": "VITA",
      "faculty": "ENAC",
      "mdh_focus": [],
      "dataTypes": [
        "Image",
        "Video"
      ],
      "techTypes": [
        "Computer Vision",
        "Social Behaviour Prediction",
        "Multi-agent Modelling"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "odobez",
      "name": "Jean-Marc Odobez",
      "url": "https://people.epfl.ch/jean-marc.odobez",
      "unit": "Perception and Activity Understanding Group",
      "faculty": "External (Idiap; EPFL adjunct faculty)",
      "mdh_focus": [],
      "dataTypes": [
        "Video",
        "Image",
        "Audio"
      ],
      "techTypes": [
        "Computer Vision",
        "Multimodal ML",
        "Behaviour Analysis"
      ],
      "stage": "Monitoring",
      "publications": []
    },
    {
      "id": "irgc",
      "name": "EPFL International Risk Governance Center (IRGC)",
      "unit": "IRGC",
      "faculty": "EPFL",
      "mdh_focus": [
        "M",
        "D"
      ],
      "dataTypes": [],
      "techTypes": [
        "Risk Governance",
        "Deepfake Governance",
        "Provenance Standards",
        "Platform Governance"
      ],
      "publications": [
        {
          "title": "Forged Authenticity: Governing Deepfake Risks",
          "wid": "forged-authenticity-governing-deepfake-risks",
          "type": "report",
          "year": 2020,
          "venue": "EPFL International Risk Governance Center (IRGC)",
          "link": "https://doi.org/10.5075/epfl-irgc-273296",
          "authors": [
            "Aengus Collins"
          ],
          "epfl_authors": [
            "Aengus Collins"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D",
            "H"
          ],
          "mdh_topics": [
            "Deepfakes",
            "Synthetic media",
            "Risk governance",
            "Detection",
            "Provenance",
            "Platform governance",
            "Liar's dividend"
          ],
          "stage": "Prevention + Monitoring + Mitigation",
          "relevance": 5,
          "about": "Policy report on governing the risks of deepfakes, produced from a two-day interdisciplinary expert workshop at the EPFL International Risk Governance Center with participants from policy, law, technology, academia, media and business. It explains the underlying technology and how falling computing costs have made realistic fakes cheap to produce, then proposes a way to prioritise risks along severity, scale and resilience. It highlights the liar's dividend, where the existence of deepfakes lets bad actors dismiss genuine material as fake, and sets out 15 governance responses across risk management, technology, law and society. One 2019 study counted 14678 deepfake videos online, 96 percent of them pornographic.",
          "why": "It addresses synthetic media, the erosion of trust through the liar's dividend, and a governance framework spanning detection and provenance.",
          "data": "not applicable (policy report)",
          "themes": [
            "Fraud, impersonation & forgery",
            "Media literacy & public resilience",
            "Online sexual abuse & image-based abuse",
            "Platform governance & regulation",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Deepfake pornography",
            "Harms to depicted victims",
            "Image-based sexual abuse",
            "Media provenance",
            "Technical standards"
          ],
          "key_terms": [
            "Deepfakes",
            "Risk governance",
            "Image-based sexual abuse",
            "Content provenance",
            "Deepfake detection"
          ],
          "models": [],
          "method_qualifiers": [
            "Expert elicitation"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "follow_up": [
            {
              "what": "Risk governance and the rise of deepfakes, an IRGC Spotlight on risk article (May 2021, DOI 10.5075/epfl-irgc-285637), listed by IRGC as an output of this report",
              "kind": "successor-work",
              "url": "https://dx.doi.org/10.5075/epfl-irgc-285637",
              "evidence": "On IRGC's own page for this report (irgc.org/issues/digitalisation/forged-authenticity-governing-deepfake-risks/), under the report's publication list: '<a href=\"https://dx.doi.org/10.5075/epfl-irgc-285637\">Risk governance and the rise of deepfakes</a> (Spotlight on risk article, May 2021)', sitting directly above '<a href=\"https://dx.doi.org/10.5075/epfl-irgc-273296\">Policy brief</a>', which is this report. EPFL Infoscience record 285637 confirms the article exists with the abstract: 'Deepfakes first came to prominence less than five years ago. Since then, they have surged in quantity and quality, becoming both a source of viral entertainment and of concern about the dark side of digital life. In this article, we provide a risk governance perspective on the deepfake phenomenon, arguing that it warrants greater attention. We begin by distinguishing between three levels of harm that synthetic media can lead to: individual, organisational and societal.'"
            }
          ],
          "data_description": "No empirical data (policy analysis).",
          "follow_up_checked": "2026-09-02",
          "targeted_group": [
            "Politicians",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Online sexual abuse & image-based abuse": [
              "Deepfake pornography",
              "Image-based sexual abuse",
              "Harms to depicted victims"
            ],
            "Platform governance & regulation": [
              "Deepfake legislation",
              "Technical standards"
            ],
            "Verification & content authenticity": [
              "Media provenance"
            ]
          },
          "theme_qualifiers_canonical": {
            "Online sexual abuse & image-based abuse": [
              "Deepfake pornography",
              "Harms to depicted victims"
            ],
            "Platform governance & regulation": [
              "Deepfake legislation",
              "Technical standards"
            ],
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness",
              "Media provenance"
            ]
          },
          "tentative": false
        },
        {
          "title": "Spotlight on risk: Using 'proof of personhood' to tackle social media risks",
          "wid": "using-proof-of-personhood-to-tackle-social-media-risks",
          "type": "report",
          "year": 2021,
          "venue": "EPFL International Risk Governance Center (IRGC) Spotlight on Risk series",
          "link": "https://dhcenter-unil-epfl.ch/en/2021/03/26/epfl-irgc-spotlight-on-risk-using-proof-of-personhood-to-tackle-social-media-risks/",
          "authors": [
            "Aengus Collins (EPFL IRGC)",
            "Bryan Ford (DEDIS)"
          ],
          "epfl_authors": [
            "Aengus Collins (EPFL IRGC)",
            "Bryan Ford (DEDIS)"
          ],
          "mdh_relevance": "infrastructure",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "Proof of personhood",
            "Sockpuppets",
            "Bot armies",
            "Deepfakes",
            "Social media accountability",
            "Anonymity"
          ],
          "stage": "Prevention + Mitigation",
          "relevance": 3,
          "about": "Policy brief framing the core social-media problem as the cheap, replaceable and automatable nature of fake virtual identities, which amplifies bot and sockpuppet abuse and feeds the spread of misinformation, fake news and conspiracy theories. As a low-tech response it proposes pseudonym parties: simultaneous in-person events where each attendee receives one anonymous cryptographic token per cycle, attesting that a real person showed up without revealing identifying data. Suggested uses include blocking abusive tokens rather than accounts, producing verified per-person like and follow counts, and replacing CAPTCHAs. It flags operational challenges and free-speech tensions and recommends small voluntary pilots before any scaling.",
          "why": "It frames proof of personhood as a defence against fake-account-driven misinformation, bot armies and conspiracy theories.",
          "data": "N/A (policy analysis)",
          "lab": "IRGC",
          "themes": [
            "Influence operations & coordinated manipulation",
            "Platform governance & regulation"
          ],
          "subtopics": [
            "Online anonymity",
            "Pseudonym parties",
            "Regulatory oversight & audit"
          ],
          "key_terms": [
            "Proof of personhood",
            "Pseudonym parties",
            "Online anonymity",
            "Sockpuppets",
            "Fake accounts"
          ],
          "models": [],
          "method_qualifiers": [
            "Sybil resistance"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "No empirical data (policy analysis).",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "Platform governance & regulation": [
              "Online anonymity",
              "Pseudonym parties",
              "Regulatory oversight & audit"
            ]
          },
          "theme_qualifiers_canonical": {
            "Platform governance & regulation": [
              "Online anonymity",
              "Regulatory oversight & audit"
            ]
          },
          "tentative": false
        },
        {
          "title": "Risk Governance and the Rise of Deepfakes",
          "wid": "risk-governance-and-the-rise-of-deepfakes",
          "type": "report",
          "year": 2021,
          "venue": "Policy Brief, co-authored with Aengus Collins",
          "link": "https://infoscience.epfl.ch/entities/publication/7490b999-60ed-495f-a177-0f8c2c1f33b2",
          "authors": [
            "Aengus Collins (IRGC)",
            "Touradj Ebrahimi (MMSPG)"
          ],
          "epfl_authors": [
            "Aengus Collins (IRGC)",
            "Touradj Ebrahimi (MMSPG)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D",
            "H"
          ],
          "mdh_topics": [
            "Deepfakes",
            "Risk governance",
            "Provenance",
            "C2PA",
            "AI Act",
            "Digital literacy",
            "Liar's dividend"
          ],
          "stage": "Prevention + Monitoring + Mitigation",
          "relevance": 5,
          "about": "Policy report mapping deepfake harms across individual, organizational, and societal levels and recommending a coordinated governance response to synthetic-media disinformation. It notes that deepfake videos increased tenfold between 2018 and 2020, that the winning Facebook detector reached 65 percent accuracy, and that YouTube handles 720000 hours of uploads per day, so even 99.9 percent detection accuracy is insufficient. It cites the April 2021 draft EU AI Act transparency obligations and argues that provenance approaches outperform a detection arms race. It offers 15 recommendations spanning detection, provenance, and digital literacy.",
          "why": "It maps deepfake harms across individual, organizational, and societal levels and recommends a coordinated governance response to synthetic-media disinformation.",
          "data": "not applicable (policy report)",
          "lab": "IRGC",
          "what": "Policy-facing report that looks at deepfakes through a governance lens for policymakers rather than technical researchers. It sorts the harms into three levels: individual (personal abuse, reputational damage), organizational (fraud, extortion, and exposure for activity that relies on documentary evidence), and societal (manipulation of public opinion and erosion of democratic politics). It argues that no single fix is enough and sets out 15 recommendations spanning technology (detection and content provenance verification), legal frameworks (how defamation, harassment, and copyright apply to synthetic media), and digital literacy. It also flags a central tension: encouraging skepticism toward digital content can itself erode the public trust that democratic discourse depends on. Directly relevant to synthetic-media disinformation.",
          "themes": [
            "Fraud, impersonation & forgery",
            "Media literacy & public resilience",
            "Online sexual abuse & image-based abuse",
            "Platform governance & regulation",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Deepfake pornography",
            "Digital literacy",
            "Fabricated evidence",
            "Media provenance",
            "Synthetic voice fraud"
          ],
          "key_terms": [
            "Deepfakes",
            "Media provenance",
            "Epistemic security",
            "Risk governance",
            "C2PA"
          ],
          "models": [],
          "method_qualifiers": [],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "No empirical data (policy and risk analysis).",
          "follow_up_checked": "2026-09-02",
          "targeted_group": [
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Fraud, impersonation & forgery": [
              "Fabricated evidence",
              "Synthetic voice fraud"
            ],
            "Media literacy & public resilience": [
              "Digital literacy"
            ],
            "Online sexual abuse & image-based abuse": [
              "Deepfake pornography"
            ],
            "Platform governance & regulation": [
              "Deepfake legislation"
            ],
            "Verification & content authenticity": [
              "Media provenance"
            ]
          },
          "theme_qualifiers_canonical": {
            "Fraud, impersonation & forgery": [
              "Fabricated evidence",
              "Identity impersonation",
              "Synthetic voice fraud"
            ],
            "Media literacy & public resilience": [
              "Critical thinking support",
              "Digital literacy"
            ],
            "Online sexual abuse & image-based abuse": [
              "Deepfake pornography",
              "Harms to depicted victims"
            ],
            "Platform governance & regulation": [
              "Deepfake legislation",
              "Technical standards",
              "Transparency obligations"
            ],
            "Verification & content authenticity": [
              "Deepfake detection",
              "Detector robustness",
              "Media provenance"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "imi",
      "name": "Initiative for Media Innovation (IMI)",
      "unit": "IMI",
      "faculty": "EPFL",
      "mdh_focus": [
        "M",
        "D"
      ],
      "dataTypes": [],
      "techTypes": [
        "Media Literacy",
        "Fact-checking",
        "AI and Journalism",
        "Digital Trust"
      ],
      "publications": []
    },
    {
      "id": "epfl-other",
      "name": "Other EPFL contributions",
      "unit": "Various",
      "faculty": "EPFL",
      "dataTypes": [],
      "techTypes": [],
      "publications": [
        {
          "title": "LLM Detectors",
          "wid": "chapter-22-llm-detectors",
          "type": "publication",
          "year": 2024,
          "venue": "Large Language Models in Cybersecurity (eds. A. Kucharavy et al.), Springer, Chapter 22, pp. 197-204",
          "link": "https://doi.org/10.1007/978-3-031-54827-7_22",
          "authors": [
            "Henrique Da Silva Gameiro"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "AI-generated content",
            "Detection",
            "Adversarial robustness",
            "Watermarking"
          ],
          "stage": "Monitoring + Prevention",
          "relevance": 3,
          "about": "Survey of tools that try to detect text written by language models. It separates general detectors, aimed at broad uses such as spotting misinformation or propaganda, from specific detectors targeting one content type such as hate speech or spam, reviews how they are built and attacked, and covers mitigations including watermarking and retrieval-based defences. It reports that detection is fragile: paraphrasing attacks break essentially every current defence, almost all detectors work only in English, and false positives fall disproportionately on neurodivergent people and non-native English speakers. OpenAI's general classifier reports a 26 percent true positive rate, and GPTZero's roughly 10 percent detection accuracy can drop to about 1 percent under a paraphrasing attack.",
          "why": "It assesses detection of AI-generated text used for misinformation, propaganda and hate speech, and shows these detectors are easily evaded.",
          "data": "Survey of published detector evaluations (OpenAI classifier, GPTZero, DetectGPT, watermarking and retrieval-based schemes); reuses results from the author's own arXiv experiment (Stochastic Parrots Looking for Stochastic Parrots, arXiv 2304.08968)",
          "lab": "Henrique Da Silva Gameiro",
          "themes": [
            "Verification & content authenticity"
          ],
          "subtopics": [
            "AI-generated text detection",
            "Detector robustness",
            "Media provenance"
          ],
          "key_terms": [
            "Paraphrasing attacks",
            "LLM-generated text detection",
            "Watermarking",
            "Fake news",
            "Academic dishonesty"
          ],
          "models": [
            "BERT",
            "BLOOM",
            "ChatGPT",
            "DetectGPT",
            "GPT-3",
            "GPT-4",
            "GPT-NeoX",
            "GPTZero",
            "LLaMA",
            "T5"
          ],
          "method_qualifiers": [
            "Adversarial evasion",
            "Paraphrasing attacks"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "No empirical data (survey chapter).",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "Verification & content authenticity": [
              "Detector robustness",
              "AI-generated text detection",
              "Media provenance"
            ]
          },
          "theme_qualifiers_canonical": {
            "Verification & content authenticity": [
              "AI-generated text detection",
              "Detector robustness",
              "Media provenance"
            ]
          },
          "tentative": false
        },
        {
          "title": "Decoding the Discourse: Analyzing the Linguistic Features and Strategies Behind the Querdenken Movement's COVID-19 Narrative",
          "wid": "decoding-the-discourse-analyzing-the-linguistic-feature",
          "type": "publication",
          "year": 2025,
          "venue": "Health Communication, 40:12, 2591-2601",
          "link": "https://doi.org/10.1080/10410236.2025.2469936",
          "authors": [
            "Alexander Sobieska (TU Munich)",
            "Max Hampel (TU Munich)",
            "Rosa Weidenspointner (TU Munich)",
            "Valentin Pauli (TU Munich)",
            "Cheng Pan (CREATE Lab)",
            "Seong-Min Jun (TU Munich)",
            "Pia Gutsmiedl (TU Munich)"
          ],
          "epfl_authors": [
            "Cheng Pan (CREATE Lab)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "COVID-19",
            "Health misinformation",
            "Discourse analysis",
            "Alternative media",
            "Conspiracy narratives"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "Critical discourse analysis of how the German-speaking Querdenken movement's alternative media framed COVID-19. It compares the language of five Querdenken-related outlets against broadsheet and tabloid coverage, using dictionary-based automated analysis and a manual check of how the outlets handled their cited sources. The Querdenken outlets carried a more negative emotional tone, used more health-related and research-related language, and frequently misrepresented their sources, often dropping or adding information or contradicting the original authors. The study calls for stronger scientific literacy and health communication to counter the resulting health misinformation.",
          "why": "It empirically documents how alternative-media outlets manufacture health misinformation by co-opting scientific language and misrepresenting sources.",
          "data": "Web-scraped corpus of 25934 COVID-19 articles from five Querdenken-related outlets (Report24, Uncut-News, Rubikon, Transition News, Corona Blog) plus 3241 broadsheet (Der Tagesspiegel) and tabloid (Bild) articles via LexisNexis (March 2020 to December 2021); a German scientific-language dictionary; German LIWC and linear mixed models, plus a manual content analysis of over 50 randomly selected claims per outlet",
          "lab": "CREATE Lab",
          "themes": [
            "Media framing & narrative analysis",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Coverage tone",
            "Issue framing",
            "Rhetorical legitimation",
            "Selective interpretation",
            "Source referencing practices"
          ],
          "key_terms": [
            "Querdenken movement",
            "Alternative media",
            "Source misrepresentation",
            "COVID-19 misinformation",
            "COVID-19"
          ],
          "models": [],
          "method_qualifiers": [
            "Critical discourse analysis",
            "Linear mixed models"
          ],
          "events_cases": [
            "COVID-19 pandemic",
            "Querdenken movement"
          ],
          "built_at_epfl": [
            {
              "name": "German-language dictionary for the usage of scientific language",
              "kind": "dataset",
              "evidence": "We decided not to make the dataset publicly available to prevent the spread of potentially harmful information; however, our datasets are available from the corresponding author upon reasonable request."
            }
          ],
          "data_description": "25934 Querdenken articles and 3241 newspaper articles, 2020-2021.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "Germany"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Media framing & narrative analysis": [
              "Coverage tone",
              "Rhetorical legitimation",
              "Issue framing"
            ],
            "Verification & content authenticity": [
              "Source referencing practices",
              "Selective interpretation"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media framing & narrative analysis": [
              "Coverage tone",
              "Issue framing",
              "Rhetorical legitimation"
            ],
            "Verification & content authenticity": [
              "Source referencing practices"
            ]
          },
          "tentative": false
        },
        {
          "title": "Self-Moderation in the Decentralized Era: Decoding Blocking Behavior on Bluesky",
          "wid": "self-moderation-in-the-decentralized-era-decoding-block",
          "type": "publication",
          "year": 2026,
          "venue": "ICWSM 2026",
          "link": "https://infoscience.epfl.ch/handle/20.500.14299/263970",
          "authors": [
            "Carlo Alberto Bono",
            "Nick Liu",
            "Giuseppe Russo",
            "Francesco Pierri"
          ],
          "epfl_authors": [
            "Giuseppe Russo"
          ],
          "about": "Study of blocking on Bluesky, a decentralized platform where blocks are public, using more than 100M actions by nearly 2M users over three months in 2024. For 427118 users with at least 10 posts, it builds 86 features on activity, content and toxicity, shared news domains and network position, and estimates users' propensity to be blocked through classification and regression. Classifiers using all features reach maximum AUC values of 0.892 (raw block counts) and 0.875 (activity-normalized), and as few as four features achieve near-maximum performance at higher thresholds. Blocked users tend to be more active and moderately more toxic, but blocking does not appear systematically linked to sharing low-credibility content.",
          "themes": [
            "Content moderation & enforcement",
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Identity attack",
            "Self-moderation",
            "Toxicity prediction",
            "User blocking"
          ],
          "mdh_focus": [
            "H"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 3,
          "stage": "Mitigation",
          "key_terms": [
            "User blocking",
            "Self-moderation",
            "Decentralized social media",
            "Bluesky",
            "AT Protocol"
          ],
          "models": [
            "AutoGluon",
            "Detoxify",
            "XGBoost"
          ],
          "method_qualifiers": [
            "Explainability & interpretability",
            "Feature ablation",
            "Supervised classification"
          ],
          "events_cases": [
            "2024 U.S. Presidential elections"
          ],
          "built_at_epfl": [],
          "data_description": "100M+ blocking actions, 2M users, 3 months of Bluesky Firehose data.",
          "platform": [
            "Bluesky"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Self-moderation",
              "User blocking"
            ],
            "Toxicity & harassment": [
              "Identity attack",
              "Toxicity prediction"
            ]
          },
          "theme_qualifiers_canonical": {},
          "tentative": false
        }
      ]
    },
    {
      "id": "tugrulcan-elmas",
      "name": "Tuğrulcan Elmas",
      "unit": "University of Edinburgh",
      "faculty": "formerly LSIR, EPFL (PhD 2022), has published on MDH topics since leaving EPFL",
      "mdh_focus": [],
      "dataTypes": [],
      "techTypes": [],
      "stage": "Monitoring + Mitigation",
      "publications": [
        {
          "title": "Tactical Reframing of Online Disinformation Campaigns Against The Istanbul Convention",
          "wid": "tactical-reframing-of-online-disinformation-campaigns-a",
          "type": "publication",
          "year": 2021,
          "venue": "ICWSM 2021 Workshop on Data Mining for Online Misinformation and Disinformation (DWMV)",
          "link": "https://arxiv.org/abs/2105.13398",
          "authors": [
            "Tugrulcan Elmas (LSIR)",
            "Rebekah Overdorf (EPFL)",
            "Karl Aberer (LSIR)"
          ],
          "epfl_authors": [
            "Tugrulcan Elmas (LSIR)",
            "Rebekah Overdorf (EPFL)",
            "Karl Aberer (LSIR)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D",
            "H"
          ],
          "mdh_topics": [
            "Disinformation campaigns",
            "Narrative reframing",
            "Homophobia",
            "Gender-based violence",
            "Turkey",
            "Facebook",
            "Tactical reframing",
            "Astroturfing",
            "Cross-actor coordination"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "Empirical study tracing how an online disinformation campaign in Turkey shifted its message to build support for leaving the Istanbul Convention, the human-rights treaty on violence against women. Using public Facebook posts, it shows the campaign began in divorced men's groups complaining about the domestic implementing law, then was reframed to attack the convention by stressing its recognition of sexual orientation and non-traditional gender roles. Small men's-rights groups had their content amplified by larger political and religious pages and by a pro-government newspaper that shifted its own coverage the same way. It is presented as the first case study of narrative reframing inside a social-media disinformation campaign.",
          "why": "It documents a disinformation campaign that fused false framing with homophobic hate to roll back women's rights.",
          "data": "Text, Graph/Network (CrowdTangle public posts from ~2500 Turkish Facebook groups, pages, profiles)",
          "themes": [
            "Gender-based violence & misogyny",
            "Influence operations & coordinated manipulation",
            "Media framing & narrative analysis"
          ],
          "subtopics": [
            "Anti-gender campaigns",
            "Men's rights groups",
            "Narrative reframing",
            "Weaponised homophobia"
          ],
          "key_terms": [
            "Istanbul Convention",
            "Tactical reframing",
            "Homophobia",
            "Men's rights groups",
            "Turkey"
          ],
          "models": [],
          "method_qualifiers": [
            "Frame analysis",
            "Manual annotation",
            "Retrospective archive mining"
          ],
          "events_cases": [
            "Turkey's withdrawal from the Istanbul Convention"
          ],
          "built_at_epfl": [],
          "data_description": "Facebook posts from 2500 tracked Turkish groups, pages and profiles, 2014-2020.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Facebook"
          ],
          "region_country": [
            "Turkey"
          ],
          "targeted_group": [
            "LGBTQ+",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Gender-based violence & misogyny": [
              "Weaponised homophobia",
              "Anti-gender campaigns",
              "Men's rights groups"
            ],
            "Influence operations & coordinated manipulation": [
              "Grassroots campaign organizing",
              "Narrative reframing"
            ],
            "Media framing & narrative analysis": [
              "Narrative shift over time"
            ]
          },
          "theme_qualifiers_canonical": {
            "Gender-based violence & misogyny": [
              "Anti-gender campaigns",
              "Men's rights groups",
              "Weaponised homophobia"
            ],
            "Influence operations & coordinated manipulation": [
              "Grassroots campaign organizing",
              "Narrative reframing"
            ],
            "Media framing & narrative analysis": [
              "Issue framing",
              "Narrative shift over time"
            ]
          },
          "tentative": false
        },
        {
          "title": "The Role of Compromised Accounts in Social Media Manipulation (Doctoral Thesis)",
          "wid": "the-role-of-compromised-accounts-in-social-media-manipu",
          "type": "thesis",
          "year": 2022,
          "venue": "EPFL Doctoral Thesis (Thèse n° 8991), IC Faculty, LSIR",
          "link": "https://infoscience.epfl.ch/record/297318",
          "authors": [
            "Tuğrulcan Elmas"
          ],
          "epfl_authors": [
            "Tuğrulcan Elmas"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D",
            "H"
          ],
          "mdh_topics": [
            "Social media manipulation",
            "Compromised accounts",
            "Astroturfing",
            "Bot detection",
            "Twitter trends",
            "Account repurposing"
          ],
          "stage": "Monitoring + Mitigation",
          "relevance": 5,
          "about": "Doctoral thesis on how attackers weaponise compromised social-media accounts, structured around three contributions. It introduces ephemeral astroturfing, an attack that pushes a keyword or trend then deletes the activity so accounts can be reused. It shows that retweet bots bought on black markets are compromised real accounts rather than purpose-built ones, challenging prior bot-detection assumptions. It also builds a pipeline that finds accounts whose identity was changed to repurpose them while keeping their followers. The work detected over 19000 fake Twitter trends promoted by more than 108000 accounts.",
          "why": "It characterises how compromised accounts drive trend manipulation and disinformation campaigns.",
          "data": "Text, Graph/Network",
          "themes": [
            "Content moderation & enforcement",
            "Economics & incentives of MDH",
            "Influence operations & coordinated manipulation",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Account suspension analysis",
            "Fake trending topics",
            "Moderator-assist detection",
            "News-feed suppression",
            "Popularity mechanism manipulation"
          ],
          "key_terms": [
            "Ephemeral astroturfing",
            "Compromised accounts",
            "Retweet bots",
            "Misleading repurposing",
            "Astroturfing"
          ],
          "models": [
            "BERT"
          ],
          "method_qualifiers": [
            "Anomaly detection",
            "Supervised classification",
            "Honeypot infiltration"
          ],
          "events_cases": [
            "#SuriyelilerDefolsun campaign",
            "2016 U.S. election Russian interference"
          ],
          "built_at_epfl": [
            {
              "name": "Astrobot dataset",
              "kind": "dataset"
            },
            {
              "name": "Ephemeral astroturfing dataset",
              "kind": "dataset"
            },
            {
              "name": "EphemeralAstroturfing",
              "kind": "dataset"
            },
            {
              "name": "Misleading repurposing dataset",
              "kind": "dataset"
            },
            {
              "name": "Real-time fake trend detection Twitter bot",
              "kind": "tool"
            },
            {
              "name": "Retweet bot dataset",
              "kind": "dataset",
              "url": "https://github.com/tugrulz/RetweetBots",
              "evidence": "Thesis, Chapter 4 (Retweet Bots): \"The datasets are made available for reproducibility 1\", footnote \"1 https://github.com/tugrulz/RetweetBots\". Repository README: \"This repository contains the data described in Characterizing Retweet Bots: The Case of Black Market Accounts in dataset.csv\", split into the timeline dataset (\"These bots are not suspended (but probably not active), so you can readily collect them using Twitter's statuses/user_timeline endpoint.\") and the archive dataset (\"These are suspended, you need to collect them from Internet archive's Twitter dataset.\")."
            },
            {
              "name": "RetweetBots",
              "kind": "dataset"
            },
            {
              "name": "WayPop Machine",
              "kind": "tool",
              "url": "https://github.com/tugrulz/WayPop",
              "evidence": "Repository description: \"WayPop: A Wayback Machine to Investigate Popularity and Root Out Trolls\", forked from LSIR/Twitter-Time-Machine (LSIR being Karl Aberer's EPFL lab, the thesis's host lab). README: \"This repository contains the code to run the website for our application Twitter Time Machine, as well as the code to generate the data.\" and \"This application was created as part of our semester project at EPFL with LSIR. ... Developed by Thomas Ibanez & Alexandre Hutter.\" Thesis section 5.11 is titled \"WayPop Machine: A Wayback Machine to Investigate Repurposed Accounts\" and its Figure 5.8 caption describes exactly this stack: \"the processed data are stored in a NoSQL database, MongoDB. The web server built using the Django framework communicates with the data layer\". The two named developers are co-authors of the corresponding paper, Elmas, Ibanez, Hutter, Overdorf, Aberer, FOSINT-SI/ASONAM 2022."
            },
            {
              "name": "Ephemeral astroturfing attack dataset",
              "kind": "dataset",
              "url": "https://github.com/tugrulz/EphemeralAstroturfing",
              "evidence": "Thesis, Chapter 3 (Ephemeral Astroturfing), reproducibility statement: \"This research was conducted using the Internet Archive's Twitter Stream Grab and trends data, so all data is public. and the study is reproducible. In addition, the IDs of the tweets and users annotated in this study as well as the annotated attacks are made available 2\", footnote \"2 https://github.com/tugrulz/EphemeralAstroturfing\". Repository README: \"This repository contains the data, the annotations and the code for the paper 'Analyzing Activity and Suspension Patterns of Twitter Bots Attacking Turkish Twitter Trends by a Longitudinal Dataset' and 'Ephemeral Astroturfing Attacks: The Case of Fake Twitter Trends'.\" It ships fake_trends.csv, astrobot_annotations.csv and attack_annotations.csv."
            },
            {
              "name": "Repurposed accounts ground-truth dataset",
              "kind": "dataset",
              "url": "https://github.com/tugrulz/MisleadingRepurposing",
              "evidence": "Chapter 5 of the thesis is the paper 'Misleading Repurposing on Twitter' (Elmas, Overdorf, Aberer), whose published abstract ends: \"The data and the code is available at https://github.com/tugrulz/MisleadingRepurposing.\" The thesis states the artefact as contribution 3: \"establish a hand-labeled ground-truth dataset of repurposed accounts using datasets published by Twitter\"."
            }
          ],
          "follow_up": [
            {
              "what": "Chapter 5 of the thesis was published as 'Misleading Repurposing on Twitter' at ICWSM 2023 (Elmas, Overdorf, Aberer), a year after the thesis was defended, and that publication is what makes the ground-truth dataset and code public",
              "kind": "successor-work",
              "url": "https://ojs.aaai.org/index.php/ICWSM/article/view/22139",
              "evidence": "Article record: \"Misleading Repurposing on Twitter\", authors \"Tugrulcan Elmas, Rebekah Overdorf, Karl Aberer\", \"Vol. 17 (2023)\", published 2 June 2023. Abstract: \"We present the first in-depth and large-scale study of misleading repurposing ... We found over 100000 accounts that may have been repurposed. Of those, 28% were removed from the platform after 2 years, thereby confirming their inauthenticity. ... The data and the code is available at https://github.com/tugrulz/MisleadingRepurposing.\" The thesis version (2022) reports the 100000 figure but not the 28% two-year removal confirmation."
            },
            {
              "what": "Longitudinal astrobot dataset extending Chapter 3's ephemeral astroturfing detection from the thesis's 2019 annotation window to 212000+ bots and 29000 fake trends over 2015-2022, released into the thesis's own EphemeralAstroturfing repository (Elmas, WWW 2023 Companion)",
              "kind": "dataset",
              "url": "https://arxiv.org/abs/2304.07907",
              "evidence": "Abstract: \"Past work on such fake trends revealed a new astroturfing attack named ephemeral astroturfing that employs a very unique bot behavior in which bots post and delete generated tweets in a coordinated manner. As such, it is easy to mass-annotate such bots reliably, making them a convenient source of ground truth for bot research. In this paper, we detect and disclose over 212000 such bots targeting Turkish trends, which we name astrobots. ... We found that Twitter purged those bots en-masse 6 times since June 2018. ... The dataset is publicly available at https://github.com/tugrulz/EphemeralAstroturfing.\""
            }
          ],
          "data_description": "1% Twitter Stream Grab: 19485 fake trends, 108682 astrobots, 6199 retweet bots.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter"
          ],
          "region_country": [
            "Turkey",
            "United States"
          ],
          "targeted_group": [
            "LGBTQ+",
            "Syrian refugees",
            "migrants"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Moderator-assist detection",
              "News-feed suppression",
              "Account suspension analysis"
            ],
            "Influence operations & coordinated manipulation": [
              "Compromised account botnets",
              "Fake trending topics"
            ],
            "Spread, amplification & networks": [
              "Popularity mechanism manipulation"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Deplatforming",
              "Moderator-assist detection",
              "News-feed suppression"
            ],
            "Influence operations & coordinated manipulation": [
              "Compromised account botnets",
              "Ephemeral astroturfing"
            ],
            "Spread, amplification & networks": [
              "Account coordination graphs",
              "Popularity mechanism manipulation"
            ]
          },
          "tentative": false
        },
        {
          "title": "Can Celebrities Burst Your Bubble?",
          "wid": "can-celebrities-burst-your-bubble",
          "type": "publication",
          "year": 2020,
          "venue": "arXiv 2003.06857; WWW 2020 Companion (Innovative Ideas in Data Science workshop)",
          "link": "https://arxiv.org/abs/2003.06857",
          "authors": [
            "Tuğrulcan Elmas",
            "Kristina Hardi",
            "Rebekah Overdorf",
            "Karl Aberer"
          ],
          "epfl_authors": [
            "Tuğrulcan Elmas",
            "Kristina Hardi",
            "Rebekah Overdorf",
            "Karl Aberer"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Filter bubbles",
            "Polarization",
            "Random Walk Controversy",
            "Celebrities-as-bridges",
            "Echo chambers",
            "Counter-narrative"
          ],
          "stage": "Mitigation",
          "relevance": 2,
          "about": "A short paper proposing celebrities as a way to reduce online political polarization. The idea is to recommend polarizing or contrarian topics to celebrities so that, when they weigh in, their followers on both sides of a debate are exposed to opposing viewpoints. It frames the choice of which accounts to involve as finding ones that are both popular and politically neutral, and tests this on a Turkish election case study. Adding popular and neutral celebrities works far better than adding only the most popular accounts, and the effect holds even when a celebrity loses up to 80 percent of their followers.",
          "why": "It is a polarization-reduction intervention treating polarization and filter bubbles as threats to democratic discourse.",
          "data": "Text + Graph (Twitter follower data); case study on 2019 Istanbul Election Rerun + 81 Turkish celebrities; #Russia_March topic for empirical polarization model",
          "themes": [
            "Persuasion & cognitive effects",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Echo chambers",
            "Political polarisation",
            "Spread interventions"
          ],
          "key_terms": [
            "Filter bubbles",
            "Celebrity influence",
            "Polarization",
            "Echo chambers",
            "2019 Istanbul election rerun"
          ],
          "models": [],
          "method_qualifiers": [
            "Random walk simulation",
            "Manual annotation"
          ],
          "events_cases": [
            "#Russia_March Twitter debate",
            "2019 Istanbul Election Rerun"
          ],
          "built_at_epfl": [],
          "data_description": "Twitter follower graphs and 679 replies to 81 celebrity tweets, Istanbul 2019.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "region_country": [
            "Russia",
            "Turkey"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Spread, amplification & networks": [
              "Echo chambers",
              "Spread interventions",
              "Political polarisation"
            ]
          },
          "theme_qualifiers_canonical": {
            "Spread, amplification & networks": [
              "Echo chambers",
              "Political polarisation",
              "Spread interventions"
            ]
          },
          "tentative": false
        },
        {
          "title": "WayPop Machine: A Wayback Machine to Investigate Popularity and Root Out Trolls",
          "wid": "waypop-machine-a-wayback-machine-to-investigate-popular",
          "type": "publication",
          "year": 2022,
          "venue": "ASONAM 2022 (IEEE/ACM Int. Conf. on Advances in Social Networks Analysis and Mining)",
          "link": "https://www.computer.org/csdl/proceedings-article/asonam/2022/10068665/1LKx6Psx6ve",
          "authors": [
            "Tugrulcan Elmas",
            "Thomas Romain Ibanez",
            "Alexandre Hutter",
            "Rebekah Overdorf",
            "Karl Aberer"
          ],
          "epfl_authors": [
            "Tugrulcan Elmas",
            "Thomas Romain Ibanez",
            "Alexandre Hutter",
            "Rebekah Overdorf",
            "Karl Aberer"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Popularity manipulation",
            "Fake amplification",
            "Trolls",
            "Bot detection",
            "Wayback Machine",
            "Social media manipulation"
          ],
          "stage": "Monitoring",
          "relevance": 4,
          "about": "A method and open-source tool that works out why a social-media account became popular. The motivation is that malicious users such as trolls must manufacture their popularity on the platform itself, often illicitly through fake amplification, unlike celebrities whose fame comes from offline activity. By reconstructing an account's follower and popularity history through the Wayback Machine, the tool helps tell apart accounts that grew honestly from those whose influence was bought or faked.",
          "why": "It detects inauthentically amplified accounts and trolls, supporting disinformation investigation.",
          "data": "Text, social-media follower/popularity histories via the Wayback Machine",
          "themes": [
            "Influence operations & coordinated manipulation",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Account repurposing",
            "Influential-user tracking",
            "Information cascades",
            "Popularity mechanism manipulation"
          ],
          "key_terms": [
            "OSINT",
            "Misleading repurposing",
            "Follower growth",
            "Trolls",
            "Twitter"
          ],
          "models": [],
          "method_qualifiers": [
            "Anomaly detection",
            "Retrospective archive mining"
          ],
          "events_cases": [
            "TDP Twitter account repurposing",
            "TDP account compromise"
          ],
          "built_at_epfl": [
            {
              "name": "WayPop",
              "kind": "tool",
              "url": "https://github.com/tugrulz/WayPop",
              "evidence": "Paper: \"WayPop is publicly available on GitHub at https://github.com/tugrulz/WayPop.\" The repository description reads \"A Wayback Machine to Investigate Popularity and Root Out Trolls\", and its README states \"This application was created as part of our semester project at EPFL with LSIR\" and credits \"Thomas Ibanez & Alexandre Hutter\", two of the paper's co-authors."
            }
          ],
          "data_description": "4.67 TB Twitter Stream Grab, 1% sample, popular users >5k followers.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "region_country": [
            "Turkey",
            "United States"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Influence operations & coordinated manipulation": [
              "Account repurposing"
            ],
            "Spread, amplification & networks": [
              "Popularity mechanism manipulation",
              "Influential-user tracking",
              "Information cascades"
            ]
          },
          "theme_qualifiers_canonical": {
            "Influence operations & coordinated manipulation": [
              "Account repurposing"
            ],
            "Spread, amplification & networks": [
              "Influential-user tracking",
              "Information cascades",
              "Popularity mechanism manipulation"
            ]
          },
          "tentative": false
        },
        {
          "title": "Characterizing Retweet Bots: The Case of Black Market Accounts",
          "wid": "characterizing-retweet-bots-the-case-of-black-market-ac",
          "type": "publication",
          "year": 2022,
          "venue": "ICWSM 2022 (Sixteenth International AAAI Conference on Web and Social Media); arXiv 2112.02366",
          "link": "https://arxiv.org/abs/2112.02366",
          "authors": [
            "Tuğrulcan Elmas",
            "Rebekah Overdorf",
            "Karl Aberer"
          ],
          "epfl_authors": [
            "Tuğrulcan Elmas",
            "Rebekah Overdorf",
            "Karl Aberer"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Retweet bots",
            "Black-market accounts",
            "Compromised accounts",
            "Bot detection challenges",
            "Twitter",
            "Inauthentic amplification"
          ],
          "stage": "Monitoring",
          "relevance": 4,
          "about": "The first study to focus specifically on retweet bots, the accounts paid to amplify tweets. Instead of relying on error-prone human labelling, the authors obtained reliable examples by directly purchasing retweet services from black-market vendors, then characterised how the bots behave over their lifecycle against human-annotated controls. A central finding is that many retweet bots are not freshly mass-created accounts but compromised genuine accounts driven aggressively by an attacker, so several assumptions in earlier bot-detection work do not hold up.",
          "why": "It is a bot-detection study on inauthentic amplification that overturns prior assumptions in disinformation work.",
          "data": "Text + Account metadata (Twitter); 862 non-suspended retweet bots + 5332 suspended bots purchased from black market (extending Golbeck 2019); control groups of 27622 human-annotated accounts; 1.2M retweets + 126K tweets in timeline dataset; 302K retweets + 30K tweets in Internet Archive Stream Grab dataset",
          "themes": [
            "Economics & incentives of MDH",
            "Influence operations & coordinated manipulation",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Black-market accounts",
            "Bot amplification",
            "Purchased engagement",
            "Retweet timing patterns"
          ],
          "key_terms": [
            "Retweet bots",
            "Account compromise",
            "Black market accounts",
            "Astroturfing",
            "Bot detection"
          ],
          "models": [],
          "method_qualifiers": [
            "Wordshift analysis",
            "Welch's t-test",
            "Retrospective archive mining"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "RetweetBots dataset",
              "kind": "dataset"
            },
            {
              "name": "RetweetBots",
              "kind": "dataset",
              "url": "https://github.com/tugrulz/RetweetBots",
              "evidence": "From the paper's own full text: 'The datasets are made available for reproducibility 1 .' with footnote '1 https://github.com/tugrulz/RetweetBots'. The repository exists and its readme cites 'Characterizing Retweet Bots: The Case of Black Market Accounts' (Elmas, Overdorf, Aberer, ICWSM 2022); it holds a CSV of account ids split into the timeline subset (active or unsuspended accounts, collectable via Twitter statuses/user_timeline) and the archive subset (suspended accounts, via the Internet Archive Twitter Stream Grab), with full data available on request for research use."
            }
          ],
          "data_description": "6199 retweet-bot accounts and 27622 human accounts, about 1.5M retweets.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Economics & incentives of MDH": [
              "Black-market accounts",
              "Purchased engagement"
            ],
            "Influence operations & coordinated manipulation": [
              "Bot amplification",
              "Compromised account botnets",
              "Black-market retweets"
            ],
            "Spread, amplification & networks": [
              "Retweet timing patterns"
            ]
          },
          "theme_qualifiers_canonical": {
            "Economics & incentives of MDH": [
              "Black-market accounts",
              "Purchased engagement"
            ],
            "Influence operations & coordinated manipulation": [
              "Bot amplification",
              "Compromised account botnets"
            ],
            "Spread, amplification & networks": [
              "Account coordination graphs",
              "Popularity mechanism manipulation"
            ]
          },
          "tentative": false
        },
        {
          "title": "Ephemeral Astroturfing Attacks: The Case of Fake Twitter Trends",
          "wid": "ephemeral-astroturfing-attacks-the-case-of-fake-twitter",
          "type": "publication",
          "year": 2021,
          "venue": "IEEE European Symposium on Security and Privacy (EuroS&P) 2021",
          "link": "https://doi.org/10.1109/EuroSP51992.2021.00041",
          "authors": [
            "Tuğrulcan Elmas",
            "Rebekah Overdorf",
            "Ahmed Furkan Özkalay",
            "Karl Aberer"
          ],
          "epfl_authors": [
            "Tuğrulcan Elmas",
            "Rebekah Overdorf",
            "Ahmed Furkan Özkalay",
            "Karl Aberer"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D",
            "H"
          ],
          "mdh_topics": [
            "Astroturfing",
            "Fake Twitter trends",
            "Compromised accounts",
            "Deletion-after-promotion",
            "Real-time detection"
          ],
          "stage": "Monitoring + Mitigation",
          "relevance": 5,
          "about": "A paper that discovers and characterises a manipulation technique called ephemeral astroturfing, in which coordinated inauthentic accounts push a keyword onto Twitter's trending list and then delete the activity so the compromised accounts can be reused without being flagged. Working from an archived sample of the Twitter stream, the team finds more than 19000 unique fake trends promoted by over 108000 accounts, with ephemerally astroturfed trends making up at least 20 percent of the top 10 global trends during the observation period. The authors released a real-time detector.",
          "why": "It uncovers a novel disinformation attack affecting a fifth of top global trends and ships a live detector.",
          "data": "Twitter Stream Grab (Internet Archive 1% sample); 19000+ unique fake trends; 108000+ accounts; real-time detection bot",
          "themes": [
            "Economics & incentives of MDH",
            "Influence operations & coordinated manipulation",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Black-market accounts",
            "Ephemeral astroturfing",
            "Fake follower schemes",
            "Popularity mechanism manipulation"
          ],
          "key_terms": [
            "Ephemeral astroturfing",
            "Compromised accounts",
            "Turkey",
            "Twitter trends",
            "Coordinated inauthentic behavior"
          ],
          "models": [],
          "method_qualifiers": [
            "Honeypot infiltration",
            "Decision tree",
            "Black-box auditing",
            "Community detection"
          ],
          "events_cases": [
            "#SuriyelilerDefolsun anti-refugee campaign",
            "SuriyelilerDefolsun hashtag",
            "Turkish Local Elections 2019"
          ],
          "built_at_epfl": [
            {
              "name": "Annotated ephemeral astroturfing attacks",
              "kind": "dataset"
            },
            {
              "name": "EphemeralAstroturfing",
              "kind": "dataset",
              "url": "https://github.com/tugrulz/EphemeralAstroturfing",
              "evidence": "Paper, 'Ethics and Reproducibility': \"In addition, the IDs of the tweets and users annotated in this study as well as the annotated attacks are made available 3\", footnote 3 being \"https://github.com/tugrulz/EphemeralAstroturfing\". Repository README: \"This repository contains the data, the annotations and the code for the paper ... Ephemeral Astroturfing Attacks: The Case of Fake Twitter Trends\", citing elmas2021ephemeral (Elmas, Overdorf, Özkalay, Aberer, IEEE EuroS&P 2021, pp. 403-422). It ships the annotated fake trends, tweet/user IDs with deletion timestamps, Botometer scores, and lexicon_classifier.py, \"Rule based classifier to detect lexicon tweets\"."
            }
          ],
          "follow_up": [
            {
              "what": "Longitudinal astrobot dataset extending the attack's detection method to 212000+ bots (Elmas, 'Analyzing Activity and Suspension Patterns of Twitter Bots Attacking Turkish Twitter Trends by a Longitudinal Dataset', WWW 2023 Companion), released into this paper's own repository",
              "kind": "successor-work",
              "url": "https://arxiv.org/abs/2304.07907",
              "evidence": "Abstract: \"Past work on such fake trends revealed a new astroturfing attack named ephemeral astroturfing that employs a very unique bot behavior in which bots post and delete generated tweets in a coordinated manner.\" It uses that behaviour as ground truth to detect and disclose over 212000 bots ('astrobots') targeting Turkish trends from June 2018 onward, and states: \"The dataset is publicly available at https://github.com/tugrulz/EphemeralAstroturfing.\""
            }
          ],
          "data_description": "32895 attacked trends, 108000 accounts, Twitter data 2015-2019.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "region_country": [
            "Turkey"
          ],
          "targeted_group": [
            "LGBTQ+",
            "Migrants",
            "Political candidates",
            "Refugees",
            "Syrians"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Economics & incentives of MDH": [
              "Astroturfing as a service",
              "Black-market accounts",
              "Fake follower schemes"
            ],
            "Influence operations & coordinated manipulation": [
              "Ephemeral astroturfing",
              "Compromised account botnets"
            ],
            "Spread, amplification & networks": [
              "Popularity mechanism manipulation"
            ]
          },
          "theme_qualifiers_canonical": {
            "Economics & incentives of MDH": [
              "Astroturfing as a service",
              "Black-market accounts",
              "Fake follower schemes"
            ],
            "Influence operations & coordinated manipulation": [
              "Compromised account botnets",
              "Ephemeral astroturfing"
            ],
            "Spread, amplification & networks": [
              "Account coordination graphs",
              "Information cascades",
              "Popularity mechanism manipulation"
            ]
          },
          "tentative": false
        },
        {
          "title": "A Dataset of State-Censored Tweets",
          "wid": "a-dataset-of-state-censored-tweets",
          "type": "publication",
          "year": 2021,
          "venue": "ICWSM 2021 (International AAAI Conference on Web and Social Media)",
          "link": "https://ojs.aaai.org/index.php/ICWSM/article/view/18124",
          "authors": [
            "Tuğrulcan Elmas",
            "Rebekah Overdorf",
            "Karl Aberer"
          ],
          "epfl_authors": [
            "Tuğrulcan Elmas",
            "Rebekah Overdorf",
            "Karl Aberer"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D",
            "H"
          ],
          "mdh_topics": [
            "State censorship",
            "Twitter",
            "Censored content dataset",
            "Information control"
          ],
          "stage": "Monitoring",
          "relevance": 3,
          "about": "A paper releasing a public dataset of tweets that governments asked Twitter to withhold. It exploits the fact that Twitter withholds content regionally rather than deleting it everywhere, so censored material can still be collected from outside the affected region using archived Twitter data. The authors point to uses such as studying government censorship, detecting hate speech, and measuring the effect of censorship on users.",
          "why": "It enables research on state-level information control and censorship, with hate-speech detection noted as a use case.",
          "data": "Twitter; dataset of state-censored tweets",
          "themes": [
            "Content moderation & enforcement",
            "Platform governance & regulation",
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Censorship policies",
            "Country-withheld content",
            "Legal removal requests",
            "State-ordered takedowns"
          ],
          "key_terms": [
            "Government censorship",
            "Country-withheld content",
            "Hate speech detection",
            "Twitter",
            "Dataset"
          ],
          "models": [],
          "method_qualifiers": [
            "Descriptive statistics",
            "Social network extension",
            "Retrospective archive mining"
          ],
          "events_cases": [
            "Kashmir dispute",
            "Operation Olive Branch"
          ],
          "built_at_epfl": [
            {
              "name": "A Dataset of State-Censored Tweets",
              "kind": "dataset"
            },
            {
              "name": "CensoredTweets",
              "kind": "dataset"
            },
            {
              "name": "Dataset of State-Censored Tweets",
              "kind": "dataset",
              "url": "https://doi.org/10.5281/zenodo.4439509",
              "evidence": "Paper abstract: 'The dataset is publicly available at https://doi.org/10.5281/zenodo.4439509' and Section 1: 'We made the dataset available at Zenodo: https://doi.org/10.5281/zenodo.4439509. The dataset only consists of tweet ids and user ids in order to comply with Twitter Terms of Service.' The Zenodo record is titled 'A Dataset of State-Censored Tweets' by Tugrulcan Elmas, Rebekah Overdorf and Karl Aberer (EPFL), states it 'is the dataset associated with the paper of the same name', references arxiv.org/abs/2101.05919, and was published 14 January 2021 under CC-BY-4.0."
            }
          ],
          "follow_up": [
            {
              "what": "CensoredTweets code repository, the reproduction pipeline released alongside the dataset",
              "kind": "repository",
              "url": "https://github.com/tugrulz/CensoredTweets",
              "evidence": "Paper, Section 1: 'For the documentation and the code to reproduce the pipeline please refer to https://github.com/tugrulz/CensoredTweets.' The repository README states it provides 'documentation and the code for reproduction of the paper \"A Dataset of State-Censored Tweets\"', cites 'Elmas, Tugrulcan, Overdorf, Rebekah and Aberer, Karl. \"A Dataset of State-Censored Tweets.\" arXiv preprint arXiv:2101.05919 (2021)', and links the dataset at https://zenodo.org/record/4439509."
            },
            {
              "what": "State & Geopolitical Censorship on Twitter (X): Detection & Impact Analysis of Withheld Content, CIKM 2025, by Cetinkaya and Elmas - the first quantitative impact analysis built on this dataset",
              "kind": "successor-work",
              "url": "https://arxiv.org/abs/2508.13375",
              "evidence": "From the paper's Data section: 'Censorships until 2021: The first dataset is collected using the Internet Archive's Twitter Stream Grab, which contains 1% of all tweets between September 2011 and June 2020, from which all tweets with a non-empty \"withheld in countries\" field are extracted, yielding 583437 censored tweets from 155715 unique users [8]. Fully censored accounts are identified via the Twitter User Lookup API and an inference heuristic, resulting in 4301 entirely withheld users.' Reference [8] is 'Tugrulcan Elmas, Rebekah Overdorf, and Karl Aberer. 2021. A dataset of state-censored tweets. In Proceedings of the International AAAI Conference on Web and Social Media, Vol. 15. 1009-1015.'"
            }
          ],
          "data_description": "583k censored tweets, 155k users, 22m supplemental tweets, 2012-2020.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "region_country": [
            "France",
            "Germany",
            "India",
            "Russia",
            "Turkey"
          ],
          "targeted_group": [
            "Ethnic minorities",
            "Muslims",
            "Political dissidents",
            "Religious minorities"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "State-ordered takedowns",
              "Legal removal requests",
              "Country-withheld content"
            ],
            "Platform governance & regulation": [
              "Censorship policies",
              "State-ordered takedowns",
              "Country-withheld content"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "State-ordered takedowns"
            ],
            "Platform governance & regulation": [
              "Censorship policies"
            ]
          },
          "tentative": false
        },
        {
          "title": "Misleading Repurposing on Twitter",
          "wid": "misleading-repurposing-on-twitter",
          "type": "publication",
          "year": 2023,
          "venue": "ICWSM 2023 (Seventeenth International AAAI Conference on Web and Social Media)",
          "link": "https://ojs.aaai.org/index.php/ICWSM/article/view/22139",
          "authors": [
            "Tuğrulcan Elmas (EPFL)",
            "Rebekah Overdorf (UNIL)",
            "Karl Aberer (EPFL)"
          ],
          "epfl_authors": [
            "Tuğrulcan Elmas (EPFL)",
            "Rebekah Overdorf (UNIL)",
            "Karl Aberer (EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Misleading repurposing",
            "Account identity change",
            "High-follower targeting",
            "Twitter manipulation",
            "Public detection tool"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "The first large-scale study of misleading repurposing, where someone changes an account's identity by altering profile attributes such as handle, name and bio so the account can be reused for a new purpose while keeping its existing followers. The team detects these accounts with a supervised machine-learning pipeline applied to historical Twitter data and releases a public tool to flag popular accounts that were later repurposed. It identifies more than 100000 potentially repurposed accounts, and finds adversaries tend to target accounts with large follower counts and repurpose them after a period of inactivity and tweet deletion.",
          "why": "It names and provides a detector for a concrete account-manipulation tactic used to mislead audiences.",
          "data": "Twitter Stream Grab (Internet Archive 1% sample); supervised-learning pipeline; >100000 detected repurposed accounts",
          "themes": [
            "Content moderation & enforcement",
            "Economics & incentives of MDH",
            "Fraud, impersonation & forgery",
            "Influence operations & coordinated manipulation"
          ],
          "subtopics": [
            "Account repurposing",
            "Coordinated propaganda networks"
          ],
          "key_terms": [
            "Misleading repurposing",
            "Fake account trafficking",
            "Follow-back schemes",
            "Twitter",
            "Account identity change"
          ],
          "models": [
            "BERT",
            "bert-base-multilingual-uncased"
          ],
          "method_qualifiers": [
            "Style change detection",
            "Active learning",
            "Supervised classification"
          ],
          "events_cases": [
            "2016 U.S. elections",
            "Brexit"
          ],
          "built_at_epfl": [
            {
              "name": "Misleading Repurposing dataset and classifier",
              "kind": "dataset"
            },
            {
              "name": "MisleadingRepurposing",
              "kind": "dataset",
              "url": "https://github.com/tugrulz/MisleadingRepurposing",
              "evidence": "Last line of the paper's own abstract, verbatim: \"The data and the code is available at https://github.com/tugrulz/MisleadingRepurposing.\" The same line is repeated on the publisher's record at https://ojs.aaai.org/index.php/ICWSM/article/view/22139 (DOI 10.1609/icwsm.v17i1.22139, pages 209-220), which lists the authors as Tugrulcan Elmas and Karl Aberer (EPFL) and Rebekah Overdorf (University of Lausanne)."
            },
            {
              "name": "Repurposed accounts ground-truth dataset",
              "kind": "dataset"
            }
          ],
          "data_description": "Historical Twitter profile snapshots and tweets from 446M user IDs, 2011-2020.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Influence operations & coordinated manipulation": [
              "Account repurposing",
              "Coordinated propaganda networks"
            ]
          },
          "theme_qualifiers_canonical": {
            "Influence operations & coordinated manipulation": [
              "Account repurposing",
              "Coordinated propaganda networks"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "manoel-horta-ribeiro",
      "name": "Manoel Horta Ribeiro",
      "unit": "Princeton University",
      "faculty": "formerly DLAB, EPFL (PhD 2024)",
      "mdh_focus": [],
      "dataTypes": [],
      "techTypes": [],
      "stage": "Prevention + Monitoring + Mitigation",
      "publications": [
        {
          "title": "Can online attention signals help fact-checkers fact-check?",
          "wid": "can-online-attention-signals-help-fact-checkers-fact-ch",
          "type": "publication",
          "year": 2022,
          "venue": "MEDIATE workshop at ICWSM 2022",
          "link": "https://arxiv.org/abs/2109.09322",
          "authors": [
            "Manoel Horta Ribeiro",
            "Savvas Zannettou",
            "Oana Goga",
            "Fabrício Benevenuto",
            "Robert West"
          ],
          "epfl_authors": [
            "Manoel Horta Ribeiro",
            "Robert West"
          ],
          "about": "Paper proposing a framework to study fact-checking with online attention signals, which extracts claims from fact-checks, links them with knowledge graph entities and estimates the attention these entities receive, using Google Trends. A preliminary study applying the framework to 879 COVID-19-related fact-checks done in 2020 by 81 international organizations suggests that there is often a disconnect between attention and fact-checking: in around 40 percent of countries that fact-checked ten or more claims, half or more of the ten most popular claims were not fact-checked. Claims were first fact-checked after receiving, on average, 35 percent of the total online attention they would eventually receive in 2020, with considerable variation among claims.",
          "themes": [
            "Spread, amplification & networks",
            "Verification & content authenticity"
          ],
          "subtopics": [
            "Check-worthiness ranking",
            "Cross-country claim migration",
            "Exposure intensity",
            "Information cascades"
          ],
          "mdh_focus": [
            "M"
          ],
          "mdh_relevance": "direct",
          "relevance": 5,
          "stage": "Monitoring + Mitigation",
          "key_terms": [
            "COVID-19 misinformation",
            "Google Trends",
            "Fact-checking",
            "Online attention",
            "Attention life cycle"
          ],
          "models": [],
          "method_qualifiers": [
            "Entity linking",
            "DBSCAN clustering",
            "Time-series analysis"
          ],
          "events_cases": [
            "COVID-19 pandemic"
          ],
          "built_at_epfl": [
            {
              "name": "Framework for studying fact-checking with online attention signals",
              "kind": "framework"
            }
          ],
          "data_description": "879 COVID-19 fact-checks across 72 countries, 2586 Google Trends time series.",
          "platform": [
            "Google Search"
          ],
          "region_country": [
            "Global",
            "International"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Spread, amplification & networks": [
              "Exposure intensity",
              "Information cascades",
              "Cross-country claim migration"
            ],
            "Verification & content authenticity": [
              "Check-worthiness ranking"
            ]
          },
          "theme_qualifiers_canonical": {
            "Spread, amplification & networks": [
              "Exposure intensity",
              "Information cascades"
            ],
            "Verification & content authenticity": [
              "Check-worthiness ranking"
            ]
          },
          "tentative": false
        },
        {
          "title": "The Evolution of the Manosphere Across the Web",
          "wid": "the-evolution-of-the-manosphere-across-the-web",
          "type": "publication",
          "year": 2021,
          "venue": "ICWSM 2021",
          "link": "https://arxiv.org/abs/2001.07600",
          "authors": [
            "Manoel Horta Ribeiro",
            "Jeremy Blackburn",
            "Barry Bradlyn",
            "Emiliano De Cristofaro",
            "Gianluca Stringhini",
            "Summer Long",
            "Stephanie Greenberg",
            "Savvas Zannettou"
          ],
          "epfl_authors": [
            "Manoel Horta Ribeiro"
          ],
          "about": "Large-scale characterization of the Manosphere, a conglomerate of Web-based misogynist movements focused on \"men's issues\", based on 28.8 million posts from 6 forums and 51 subreddits. Tracking activity, user migration, toxicity and misogyny over time, it finds that older communities such as Pick Up Artists and Men's Rights Activists are giving way to more extreme ones like Incels and Men Going Their Own Way, with a substantial migration of active users. The analysis suggests these newer communities are more toxic and misogynistic, and average toxicity rose from 0.2 in Incel subreddits at the /r/Incels ban to 0.3 at the start of Incels.is, a forum created hours after that ban.",
          "themes": [
            "Gender-based violence & misogyny",
            "Radicalisation & violent extremism",
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Anti-gender campaigns",
            "Comment toxicity scoring",
            "Gendered victimisation",
            "Men's rights groups",
            "Radicalisation pathways",
            "User migration"
          ],
          "mdh_focus": [
            "H"
          ],
          "mdh_relevance": "direct",
          "relevance": 5,
          "stage": "Monitoring",
          "key_terms": [
            "Manosphere",
            "Incels",
            "User migration",
            "Misogyny",
            "Toxicity"
          ],
          "models": [
            "Perspective API"
          ],
          "method_qualifiers": [
            "Jaccard similarity",
            "Lexicon-based analysis"
          ],
          "events_cases": [
            "Banning of /r/Incels (November 2017)"
          ],
          "built_at_epfl": [
            {
              "name": "Manosphere forums and subreddits dataset",
              "kind": "dataset"
            }
          ],
          "data_description": "28.8M posts from 6 forums and 51 subreddits, 2005-2019.",
          "platform": [
            "AVFM",
            "Incels.is",
            "MGTOW Forum",
            "Reddit",
            "Rooshv",
            "The Attraction"
          ],
          "targeted_group": [
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Gender-based violence & misogyny": [
              "Men's rights groups",
              "Anti-gender campaigns",
              "Gendered victimisation"
            ],
            "Radicalisation & violent extremism": [
              "Radicalisation pathways",
              "User migration"
            ],
            "Toxicity & harassment": [
              "Comment toxicity scoring"
            ]
          },
          "theme_qualifiers_canonical": {
            "Gender-based violence & misogyny": [
              "Anti-gender campaigns",
              "Gendered victimisation",
              "Men's rights groups"
            ],
            "Radicalisation & violent extremism": [
              "Radicalisation pathways",
              "User migration"
            ],
            "Toxicity & harassment": [
              "Comment toxicity scoring",
              "Online harassment of women"
            ]
          },
          "tentative": false
        },
        {
          "title": "Automated Content Moderation Increases Adherence to Community Guidelines",
          "wid": "automated-content-moderation-increases-adherence-to-com",
          "type": "publication",
          "year": 2023,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2210.10454",
          "authors": [
            "Manoel Horta Ribeiro",
            "Justin Cheng",
            "Robert West"
          ],
          "epfl_authors": [
            "Manoel Horta Ribeiro",
            "Robert West"
          ],
          "about": "Study of 412 million Facebook comments measuring how automated content moderation, enforcing community guidelines for violence and incitement, affects subsequent rule-breaking behavior and engagement. Using public comments by adult U.S. users from June to August 2022, it applies a fuzzy regression discontinuity design around the classifier score thresholds above which comments are hidden or deleted. Deleting comments decreased subsequent rule-breaking in threads with 20 or fewer comments, even among other participants, and its effect on affected users' rule-breaking outlasted its effect on their commenting, while hiding content had small and statistically insignificant effects.",
          "themes": [
            "Content moderation & enforcement"
          ],
          "subtopics": [
            "Automated pre-filtering",
            "Ban effectiveness",
            "Deplatforming"
          ],
          "mdh_focus": [
            "H"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "Mitigation",
          "key_terms": [
            "Automated content moderation",
            "Violence and incitement",
            "Rule-breaking behavior",
            "Community guidelines",
            "Anti-social behavior"
          ],
          "models": [],
          "method_qualifiers": [
            "Causal inference",
            "Fuzzy regression discontinuity"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "412M Facebook comments, 1.5M posts, 1.3M users, US, 2022.",
          "platform": [
            "Facebook"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Ban effectiveness",
              "Automated pre-filtering",
              "Deplatforming"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Automated pre-filtering",
              "Ban effectiveness",
              "Deplatforming"
            ]
          },
          "tentative": false
        },
        {
          "title": "Post Guidance for Online Communities",
          "wid": "post-guidance-for-online-communities",
          "type": "publication",
          "year": 2025,
          "venue": "CSCW 2025",
          "link": "https://arxiv.org/abs/2411.16814",
          "authors": [
            "Manoel Horta Ribeiro",
            "Robert West",
            "Ryan Lewis",
            "Sanjay Kairam"
          ],
          "epfl_authors": [
            "Manoel Horta Ribeiro",
            "Robert West"
          ],
          "about": "Randomized experiment evaluating post guidance, a moderation approach where community rules trigger interventions, such as showing a message or preventing submission, while users draft a post. Tested on Reddit with 97616 posters in 33 subreddits over 63 days, the feature increased non-removed posts by 5.8 percent, cut reports by 9.4 percent and AutoModerator removals by 34.9 percent, and raised the comments, screen views and upvotes posts received, even though fewer posts were started and submitted. It did not increase user participation, worked similarly for newcomers and veterans, and brought the biggest increases in non-removed posts to communities that set up many rules or relied heavily on AutoModerator.",
          "themes": [
            "Content moderation & enforcement"
          ],
          "subtopics": [
            "Automated pre-filtering",
            "Community-specific rules",
            "Proactive moderation"
          ],
          "mdh_focus": [
            "M",
            "H"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "Prevention",
          "key_terms": [
            "Proactive content moderation",
            "Post Guidance",
            "Moderator workload",
            "Content moderation",
            "Reddit"
          ],
          "models": [],
          "method_qualifiers": [
            "Randomized controlled trial",
            "Causal inference",
            "Field experiment",
            "Poisson regression"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Post Guidance",
              "kind": "tool"
            }
          ],
          "data_description": "97616 users, 33 subreddits, 63-day randomized field experiment.",
          "platform": [
            "Reddit"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Proactive moderation",
              "Automated pre-filtering",
              "Community-specific rules"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Automated pre-filtering"
            ]
          },
          "tentative": false
        },
        {
          "title": "Protection from Evil and Good: The Differential Effects of Page Protection on Wikipedia Article Quality",
          "wid": "protection-from-evil-and-good-the-differential-effects-",
          "type": "publication",
          "year": 2025,
          "venue": "ICWSM 2025",
          "link": "https://doi.org/10.1609/icwsm.v19i1.35896",
          "authors": [
            "Thorsten Ruprechter",
            "Manoel Horta Ribeiro",
            "Robert West",
            "Denis Helic"
          ],
          "epfl_authors": [
            "Robert West"
          ],
          "about": "Written at Princeton University, after Manoel Horta Ribeiro left EPFL. Quasi-experimental study of how page protection, which restricts who can edit an article, affects article quality on the English Wikipedia. Using decade-long data, it matches articles protected after a request for page protection with similar articles whose request was declined, and applies a difference-in-differences approach to an automated quality metric. The effect depends on the characteristics of the article before the intervention: high-quality articles are affected positively and low-quality articles negatively, and subsequent analysis suggests high-quality articles degrade when left unprotected whereas low-quality articles improve. The effect also varies across topics, with no notable effect on STEM articles.",
          "themes": [
            "Content moderation & enforcement"
          ],
          "subtopics": [
            "Ban effectiveness",
            "Moderation effect heterogeneity",
            "Page protection"
          ],
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "Mitigation",
          "key_terms": [
            "Wikipedia page protection",
            "Article quality",
            "Vandalism",
            "Page protection",
            "Content moderation"
          ],
          "models": [
            "ORES",
            "ORES articlequality",
            "ORES articletopic"
          ],
          "method_qualifiers": [
            "Difference-in-differences",
            "Propensity score matching",
            "Causal inference"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "English Wikipedia page protection and RfPP data with ORES scores",
              "kind": "dataset"
            }
          ],
          "data_description": "299k page protections, 127k requests, English Wikipedia 2012-2023",
          "platform": [
            "Wikipedia"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Page protection",
              "Moderation effect heterogeneity",
              "Ban effectiveness"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Ban effectiveness"
            ]
          },
          "tentative": false
        },
        {
          "title": "Stranger Danger! Cross-Community Interactions with Fringe Users Increase the Growth of Fringe Communities on Reddit",
          "wid": "stranger-danger-cross-community-interactions-with-fring",
          "type": "publication",
          "year": 2023,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2310.12186",
          "authors": [
            "Giuseppe Russo",
            "Manoel Horta Ribeiro",
            "Robert West"
          ],
          "epfl_authors": [
            "Manoel Horta Ribeiro",
            "Robert West"
          ],
          "about": "Study applying text-based causal inference to test whether fringe-interactions, comment exchanges between members and non-members of fringe communities, draw new members to r/Incels, r/GenderCritical and r/The_Donald on Reddit. Users who received such interactions were up to 4.2 percentage points more likely to join than similar matched users, and interactions using toxic language had a 5 percentage point higher chance of attracting newcomers than non-toxic ones. The effect varied with the communities where interactions happened, such as left or right-leaning ones; repeated for non-fringe communities, effects were smaller and not statistically significant. An estimated 7.2, 3.1 and 2.3 percent of newcomers to the three communities joined after such interactions.",
          "themes": [
            "Radicalisation & violent extremism",
            "Spread, amplification & networks",
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Comment toxicity scoring",
            "Community infiltration",
            "Radicalisation pathways",
            "User migration"
          ],
          "mdh_focus": [
            "H"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "Monitoring",
          "key_terms": [
            "Fringe communities",
            "Incels",
            "Reddit",
            "Alt-right",
            "Capitol riot"
          ],
          "models": [
            "BERT",
            "Perspective API"
          ],
          "method_qualifiers": [
            "Causal inference",
            "Propensity score matching"
          ],
          "events_cases": [
            "r/GenderCritical",
            "r/Incels",
            "r/The Donald"
          ],
          "built_at_epfl": [],
          "data_description": "Reddit comments and posts from ~15M fringe and 5M non-fringe contributions.",
          "platform": [
            "Reddit"
          ],
          "targeted_group": [
            "Trans women",
            "Transgender people"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Radicalisation & violent extremism": [
              "Radicalisation pathways",
              "User migration",
              "Community growth mechanisms"
            ],
            "Spread, amplification & networks": [
              "Community infiltration",
              "Cross-platform migration"
            ],
            "Toxicity & harassment": [
              "Comment toxicity scoring"
            ]
          },
          "theme_qualifiers_canonical": {
            "Radicalisation & violent extremism": [
              "Radicalisation pathways",
              "User migration"
            ],
            "Spread, amplification & networks": [
              "Cross-platform migration"
            ],
            "Toxicity & harassment": [
              "Comment toxicity scoring",
              "Identity-targeted hate"
            ]
          },
          "tentative": false
        },
        {
          "title": "The Amplification Paradox in Recommender Systems",
          "wid": "the-amplification-paradox-in-recommender-systems",
          "type": "publication",
          "year": 2023,
          "venue": "ICWSM 2023",
          "link": "https://arxiv.org/abs/2302.11225",
          "authors": [
            "Manoel Horta Ribeiro",
            "Veniamin Veselovsky",
            "Robert West"
          ],
          "epfl_authors": [
            "Manoel Horta Ribeiro",
            "Veniamin Veselovsky",
            "Robert West"
          ],
          "about": "Paper explaining the amplification paradox: audits found that blindly following recommendations leads users to increasingly partisan, conspiratorial or false content, yet real user traces suggest recommender systems are not the primary driver of attention toward extreme content. Simulations with a simple agent-based model of a collaborative-filtering recommender and five political topics offer a possible explanation: simulated users rarely consume niche content when given the option because it is of low utility to them, which can lead the recommender to deamplify it. The results call for a nuanced interpretation of algorithmic amplification and for modeling the utility of content to users when auditing recommender systems.",
          "themes": [
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Content nicheness",
            "Exposure intensity",
            "Recommender amplification"
          ],
          "mdh_focus": [
            "M"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "Monitoring",
          "key_terms": [
            "Algorithmic amplification",
            "Extreme content",
            "Collaborative filtering",
            "Recommender systems",
            "Agent-based model"
          ],
          "models": [],
          "method_qualifiers": [
            "Agent-based simulation",
            "Collaborative filtering"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "amplification paradox",
              "kind": "tool"
            }
          ],
          "data_description": "Synthetic simulation: 600 users, 600 items, 5 political topics.",
          "platform": [
            "YouTube"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Spread, amplification & networks": [
              "Recommender amplification",
              "Content nicheness",
              "Exposure intensity"
            ]
          },
          "theme_qualifiers_canonical": {
            "Spread, amplification & networks": [
              "Exposure intensity",
              "Recommender amplification"
            ]
          },
          "tentative": false
        },
        {
          "title": "Tube2Vec: Social and Semantic Embeddings of YouTube Channels",
          "wid": "tube2vec-social-and-semantic-embeddings-of-youtube-chan",
          "type": "publication",
          "year": 2023,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2306.17298",
          "authors": [
            "Léopaul Boesinger",
            "Manoel Horta Ribeiro",
            "Veniamin Veselovsky",
            "Robert West"
          ],
          "epfl_authors": [
            "Léopaul Boesinger",
            "Manoel Horta Ribeiro",
            "Veniamin Veselovsky",
            "Robert West"
          ],
          "about": "Paper building latent representations (embeddings) of YouTube channels as an alternative to manual annotation and low-recall keyword search when studying the social and semantic dimensions of channels. From YouTube links shared on Reddit between 2010 and 2022, it creates embeddings based on social sharing behavior, video metadata such as titles and descriptions, and YouTube's video recommendations, evaluated with crowdsourcing and existing datasets. Recommendation embeddings excel at capturing both social and semantic dimensions, although social-sharing embeddings correlate better with existing partisan scores. The embeddings for 44000 YouTube channels are shared for future research.",
          "themes": [
            "NA"
          ],
          "subtopics": [],
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_relevance": "infrastructure",
          "relevance": 2,
          "stage": "Monitoring",
          "key_terms": [
            "YouTube channel embeddings",
            "Social dimensions",
            "Recommendation graph",
            "Semantic similarity",
            "Computational social science"
          ],
          "models": [
            "all-MiniLM-L6-v2"
          ],
          "method_qualifiers": [
            "Graph embedding",
            "Sentence embeddings",
            "Random forest probing",
            "Supervised classification"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Tube2Vec embeddings",
              "kind": "dataset"
            },
            {
              "name": "YouTube channel embeddings (44K channels)",
              "kind": "dataset"
            }
          ],
          "data_description": "44000 YouTube channels, 77.4M Reddit tuples, 2010-2022.",
          "platform": [
            "Reddit",
            "YouTube"
          ],
          "label_version": "v5",
          "theme_qualifiers": {},
          "theme_qualifiers_canonical": {},
          "tentative": false
        },
        {
          "title": "On the Conversational Persuasiveness of Large Language Models: A Randomized Controlled Trial",
          "wid": "on-the-conversational-persuasiveness-of-large-language-",
          "type": "publication",
          "year": 2025,
          "link": "https://www.nature.com/articles/s41562-025-02194-6",
          "authors": [
            "Francesco Salvi (DLAB, EPFL)",
            "Manoel Horta Ribeiro (DLAB, EPFL)",
            "Riccardo Gallotti (Fondazione Bruno Kessler)",
            "Robert West (DLAB, EPFL)"
          ],
          "epfl_authors": [
            "Francesco Salvi (DLAB, EPFL)",
            "Manoel Horta Ribeiro (DLAB, EPFL)",
            "Robert West (DLAB, EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "LLM persuasion",
            "Microtargeting",
            "GPT-4",
            "Online debates",
            "Influence operations",
            "RCT"
          ],
          "stage": "Monitoring",
          "relevance": 4,
          "about": "Preregistered randomized controlled trial testing whether GPT-4 can out-persuade people in short text debates, and whether giving the model basic sociodemographic facts about its opponent makes it more effective. Participants debated either a human or GPT-4, with or without the opponent's personal data, and opinion change was measured before and after across contentious topics. When GPT-4 had personal information about its opponent, participants were far more likely to shift toward the opposing view than in human-versus-human debates; without personalization, GPT-4 had no significant edge. Personalized GPT-4 raised the odds of a participant shifting toward the opponent's position by 81.2 percent over the human-human baseline.",
          "why": "Provides experimental evidence that personalized LLMs out-persuade humans, a core mechanism of AI-driven disinformation and microtargeting.",
          "data": "Text",
          "themes": [
            "Persuasion & cognitive effects"
          ],
          "subtopics": [
            "Attitude change",
            "LLM persuasion",
            "Microtargeting"
          ],
          "key_terms": [
            "LLM persuasion",
            "Microtargeting",
            "Online debates",
            "Opinion change",
            "Personalized persuasion"
          ],
          "models": [
            "GPT-4"
          ],
          "method_qualifiers": [
            "Randomized controlled trial",
            "Causal inference",
            "Ordinal regression"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "follow_up": [
            {
              "what": "debategpt code repository (epfl-dlab), the analysis and debate-platform code for this study",
              "kind": "repository",
              "url": "https://github.com/epfl-dlab/debategpt",
              "evidence": "Code availability: \"The code to fully reproduce the analyses described in this work is available on GitHub at https://github.com/epfl-dlab/debategpt. Data collection was performed using Empirica v.1.9.5. The study was conducted using Python 3.11, R 4.3.1 and LIWC-22.\" (PMC full text of the Nature Human Behaviour version). The repository page itself states it \"contains code accompanying the research article 'On the Conversational Persuasiveness of GPT-4,' which was published in Nature Human Behaviour.\""
            },
            {
              "what": "debategpt dataset released on Hugging Face: the debate transcripts and pre/post agreement data collected in the trial",
              "kind": "dataset",
              "url": "https://huggingface.co/datasets/frasalvi/debategpt",
              "evidence": "Data availability: \"The debate dataset collected for our study is publicly available at https://huggingface.co/datasets/frasalvi/debategpt\" (PMC full text). The Hugging Face card shows 750 rows in CSV under cc-by-sa-4.0, with pre- and post-treatment agreement fields, the four treatment conditions and ~30 debate topics, matching the paper's design."
            }
          ],
          "data_description": "560 online debates involving 820 US participants.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "United States"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Persuasion & cognitive effects": [
              "LLM persuasion",
              "Microtargeting",
              "Attitude change"
            ]
          },
          "theme_qualifiers_canonical": {
            "Persuasion & cognitive effects": [
              "Attitude change",
              "LLM persuasion",
              "Microtargeting"
            ]
          },
          "tentative": false
        },
        {
          "title": "Can Language Models Recognize Convincing Arguments?",
          "wid": "can-language-models-recognize-convincing-arguments",
          "type": "publication",
          "year": 2024,
          "venue": "Findings of EMNLP 2024",
          "link": "https://aclanthology.org/2024.findings-emnlp.515/",
          "authors": [
            "Paula Dolores Rescala (EPFL)",
            "Manoel Horta Ribeiro (EPFL)",
            "Tiancheng Hu (University of Cambridge)",
            "Robert West (EPFL)"
          ],
          "epfl_authors": [
            "Paula Dolores Rescala (EPFL)",
            "Manoel Horta Ribeiro (EPFL)",
            "Robert West (EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Persuasion",
            "Microtargeting",
            "LLM evaluation",
            "Personalized misinformation",
            "Social sensing",
            "Propaganda"
          ],
          "stage": "Monitoring",
          "relevance": 4,
          "about": "Study of whether LLMs can identify which arguments are persuasive and predict how a particular person will react from their profile and prior beliefs. Several models were tested in zero-shot settings on judging which debater argued better and on predicting a person's stance before and after a debate across contentious topics. On argument quality only GPT-4 matched the human benchmark, at 60.5 percent against 60.7 percent, while the other three models scored between 24.9 and 42.7 percent against a 33.3 percent random baseline; on stance prediction every model performed on a par with crowdworkers. Because the models do well on different debates, stacking their predictions in a supervised logistic regression beats crowdworkers on the two stance tasks, though a plain XGBoost model trained on the same traits still beats every individual model and the stack.",
          "why": "It probes whether LLMs can detect persuasive, demographically tailored arguments, a capability the authors tie to personalized misinformation and propaganda.",
          "data": "Text",
          "themes": [
            "AI safety",
            "Persuasion & cognitive effects"
          ],
          "subtopics": [
            "Capability evaluation",
            "Demographic tailoring",
            "LLM persuasion",
            "Microtargeting",
            "Misuse risk assessment"
          ],
          "key_terms": [
            "LLM persuasion",
            "Political microtargeting",
            "Personalized misinformation",
            "Argument quality",
            "Argument persuasiveness"
          ],
          "models": [
            "GPT-3.5",
            "GPT-4",
            "Llama 2",
            "Mistral 7B"
          ],
          "method_qualifiers": [
            "Zero-shot prompting",
            "Model stacking"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "PoliIssues",
              "kind": "dataset",
              "url": "https://zenodo.org/records/13887286",
              "evidence": "Paper: \"Hereafter, we call this dataset the PoliIssues dataset.\" Repository README (reached from the paper's own link go.epfl.ch/persuasion-llm): \"Data and Code for: 'Can Language Models Recognize Convincing Arguments?'\" and \"Data for this work is available through Zenodo (https://zenodo.org/records/13887286).\" The Zenodo record is titled \"Can Language Models Recognize Convincing Arguments?\" by Paula Rescala and Manoel Horta Ribeiro (Ecole Polytechnique Federale de Lausanne)."
            },
            {
              "name": "PoliProp",
              "kind": "dataset",
              "url": "https://zenodo.org/records/13887286",
              "evidence": "Paper: \"Hereafter, we call this the PoliProp dataset.\" Repository README (reached from the paper's own link go.epfl.ch/persuasion-llm): \"Data and Code for: 'Can Language Models Recognize Convincing Arguments?'\" and \"Data for this work is available through Zenodo (https://zenodo.org/records/13887286).\" The Zenodo record is titled \"Can Language Models Recognize Convincing Arguments?\" by Paula Rescala and Manoel Horta Ribeiro (Ecole Polytechnique Federale de Lausanne)."
            }
          ],
          "follow_up": [
            {
              "what": "debate-gpt-x, the public data-and-code repository released with the paper (analysis notebook, the debate_gpt package, prompt/LLM-output pipeline)",
              "kind": "repository",
              "url": "https://github.com/manoelhortaribeiro/debate-gpt-x",
              "evidence": "README first line: \"# Data and Code for: 'Can Language Models Recognize Convincing Arguments?'\" and \"In order to reproduce our results, only the data in `tidy.zip` on Zenodo is needed... All the analysis is done directly in the notebook `analyses.ipynb`.\""
            }
          ],
          "data_description": "833 annotated debate.org political debates, 4871 votes, 751 crowdsourced labels.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "debate.org"
          ],
          "region_country": [
            "United States"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Misuse risk assessment",
              "Capability evaluation"
            ],
            "Persuasion & cognitive effects": [
              "Microtargeting",
              "LLM persuasion",
              "Demographic tailoring"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Misuse risk assessment"
            ],
            "Persuasion & cognitive effects": [
              "LLM persuasion",
              "Microtargeting"
            ]
          },
          "tentative": false
        },
        {
          "title": "Auditing Radicalization Pathways on YouTube",
          "wid": "auditing-radicalization-pathways-on-youtube",
          "type": "publication",
          "year": 2019,
          "link": "https://arxiv.org/abs/1908.08313",
          "authors": [
            "Manoel Horta Ribeiro (EPFL)",
            "Raphael Ottoni (UFMG)",
            "Robert West (EPFL)",
            "Virgílio A. F. Almeida (UFMG)",
            "Wagner Meira Jr. (UFMG)"
          ],
          "epfl_authors": [
            "Manoel Horta Ribeiro (EPFL)",
            "Robert West (EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "H"
          ],
          "mdh_topics": [
            "Radicalization",
            "YouTube",
            "Recommender systems",
            "Hate speech",
            "Alt-right",
            "Algorithmic audit"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "First large-scale quantitative test of the claim that YouTube acts as a radicalization pipeline moving viewers from mild contrarian content toward far-right extremism. Using videos, channels, and millions of comments sorted into mainstream media, the Intellectual Dark Web, the Alt-lite, and the white-supremacist Alt-right, it traces whether people drift from milder to more extreme content. About 12 percent of users who first commented only on Alt-lite or Intellectual Dark Web content had moved to Alt-right content within a year, three to four times the rate for users who started on mainstream media. Channel recommendations made more extreme content easy to reach from milder communities.",
          "why": "First large-scale audit linking YouTube recommendations to user migration into white-supremacist content, squarely in hate speech and radicalization.",
          "data": "Text, channel networks",
          "themes": [
            "Radicalisation & violent extremism",
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Cross-channel activity",
            "Echo chambers",
            "Far-right communities",
            "Radicalisation pathways",
            "Recommender amplification",
            "User migration"
          ],
          "key_terms": [
            "Alt-right",
            "Intellectual Dark Web",
            "Radicalization pipeline",
            "User migration",
            "Algorithmic auditing"
          ],
          "models": [],
          "method_qualifiers": [
            "Algorithmic auditing",
            "Random walk simulation",
            "Manual annotation",
            "Statistical & causal analysis"
          ],
          "events_cases": [
            "2016 US presidential election",
            "Alt-right movement",
            "YouTube radicalization"
          ],
          "built_at_epfl": [],
          "follow_up": [
            {
              "what": "Analysis code released as the GitHub repository manoelhortaribeiro/radicalization_youtube (four Jupyter notebooks reproducing the paper's figures and tables); the underlying data is withheld and shared on request",
              "kind": "repository",
              "url": "https://github.com/manoelhortaribeiro/radicalization_youtube",
              "evidence": "Code for the paper \"Auditing Radicalization Pathways on YouTube\" (FAT* 2020) ... Due to the sensitivity of the data, the data/helpers necessary to reproduce the analyses are not made available here. We are, however, willing to consider sharing it with other research groups upon request :)"
            },
            {
              "what": "The YouTube dataset was extended by the same group in \"Are Anti-Feminist Communities Gateways to the Far Right? Evidence from Reddit and YouTube\" (Mamie, Horta Ribeiro, West, WebSci 2021), which reuses the Alt-right / Alt-lite / I.D.W. channel data and collection methodology to test whether the Manosphere feeds the far right",
              "kind": "successor-work",
              "url": "https://arxiv.org/abs/2102.12837",
              "evidence": "To study YouTube, we expand the dataset obtained from (Ribeiro et al. 2020). We leverage the same methodology to collect data associated with 4 groups in the Manosphere ... Notice that we use this data along with the YouTube data published by Ribeiro et al. (Ribeiro et al. 2020), which was captured in a similar fashion."
            }
          ],
          "data_description": "330k videos, 349 channels, 72M comments, 2M recommendations.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "YouTube"
          ],
          "region_country": [
            "United States"
          ],
          "targeted_group": [
            "Ethnic minorities",
            "Religious minorities",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Radicalisation & violent extremism": [
              "Far-right communities",
              "Radicalisation pathways",
              "User migration"
            ],
            "Spread, amplification & networks": [
              "Recommender amplification",
              "Echo chambers",
              "Cross-channel activity"
            ]
          },
          "theme_qualifiers_canonical": {
            "Radicalisation & violent extremism": [
              "Far-right communities",
              "Radicalisation pathways",
              "User migration"
            ],
            "Spread, amplification & networks": [
              "Cross-platform migration",
              "Echo chambers",
              "Recommender amplification"
            ]
          },
          "tentative": false
        },
        {
          "title": "Message Distortion in Information Cascades",
          "wid": "message-distortion-in-information-cascades",
          "type": "publication",
          "year": 2019,
          "link": "https://arxiv.org/abs/1902.09197",
          "authors": [
            "Manoel Horta Ribeiro (UFMG)",
            "Kristina Gligorić (EPFL)",
            "Robert West (EPFL)"
          ],
          "epfl_authors": [
            "Kristina Gligorić (EPFL)",
            "Robert West (EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M"
          ],
          "mdh_topics": [
            "Information cascades",
            "Misinformation",
            "Science communication",
            "Summarisation",
            "Telephone effect"
          ],
          "stage": "Monitoring",
          "relevance": 4,
          "about": "Manoel Horta Ribeiro's affiliation on this paper is the Federal University of Minas Gerais (UFMG), not EPFL. Crowdsourced experiment on how information gets warped as it passes from person to person even when nobody is lying, a telephone effect where small errors accumulate over successive retellings. Workers iteratively summarized medical abstracts, either using the previous person's summary as input or always starting from the original. Cascading distorts the most important part the most: peripheral details survive, but an abstract's core conclusion is represented about 25 percentage points less often than in direct compression, and the usual advantage of domain expertise disappears.",
          "why": "It shows experimentally how accurate information becomes misinformation through cascading retelling alone, without any malicious actor.",
          "data": "Text",
          "themes": [
            "Spread, amplification & networks"
          ],
          "subtopics": [
            "Information cascades",
            "Iterative summarization",
            "Message distortion"
          ],
          "key_terms": [
            "Information cascades",
            "Message distortion",
            "Science communication",
            "Telephone effect",
            "Medical misinformation"
          ],
          "models": [],
          "method_qualifiers": [
            "Controlled experiment",
            "Crowdsourced annotation",
            "Statistical & causal analysis"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Message distortion in information cascades dataset (mdic)",
              "kind": "dataset"
            }
          ],
          "follow_up": [
            {
              "what": "epfl-dlab/mdic, the code and data repository for the paper",
              "kind": "repository",
              "url": "https://github.com/epfl-dlab/mdic",
              "evidence": "Repo README: 'This repository contains the data and code of the paper \"Message Distortion in Information Cascades\"', followed by the BibTeX entry naming Ribeiro, Gligoric and West, Proceedings of the 2019 World Wide Web Conference. The paper itself names the repo in a first-page footnote: 'Code/data: github.com/epfl-dlab/mdic'."
            },
            {
              "what": "Interactive cascade-visualisation website released alongside the data",
              "kind": "deployment",
              "url": "https://epfl-dlab.github.io/mdic/",
              "evidence": "Repo README: 'Check out the accompanying website which allows you to visualize the data.' linking to https://epfl-dlab.github.io/mdic/. The live page is titled 'Message Distortion in Information Cascades' and lists the study's medical abstracts (Breast Cancer, Immunization, ...)."
            },
            {
              "what": "The paper's MTurk iterative-summarization task was reused by the same lab to measure LLM use by crowd workers: 'Artificial Artificial Artificial Intelligence: Crowd Workers Widely Use Large Language Models for Text Production Tasks' (Veselovsky, Horta Ribeiro, West), later published in CACM 2024 as 'Prevalence and Prevention of Large Language Model Use in Crowd Work'",
              "kind": "successor-work",
              "url": "https://arxiv.org/abs/2306.07899",
              "evidence": "'We modify a prior MTurk task originally devised by Horta Ribeiro et al. 2019, whose goal was to study the so-called \"telephone effect,\" whereby information is gradually lost or distorted as a message is passed from human to human in an information cascade.' and 'In the original study, crowd workers produced eight increasingly short summaries of each original abstract, forming entire information cascades. For our purpose, however, we reduced the task to a single summarization step'."
            }
          ],
          "data_description": "Crowdsourced summaries of 16 NEJM abstracts across 5 target lengths.",
          "follow_up_checked": "2026-09-02",
          "label_version": "v5",
          "theme_qualifiers": {
            "Spread, amplification & networks": [
              "Message distortion",
              "Information cascades",
              "Iterative summarization"
            ]
          },
          "theme_qualifiers_canonical": {
            "Spread, amplification & networks": [
              "Information cascades",
              "Message distortion"
            ]
          },
          "tentative": false
        },
        {
          "title": "Analyzing the \"Sleeping Giants\" Activism Model in Brazil",
          "wid": "analyzing-the-sleeping-giants-activism-model-in-brazil",
          "type": "publication",
          "year": 2022,
          "venue": "14th ACM Web Science Conference (WebSci 2022), 87-97",
          "link": "https://arxiv.org/abs/2105.07523",
          "authors": [
            "Bárbara Gomes Ribeiro (UFMG)",
            "Manoel Horta Ribeiro (EPFL)",
            "Virgílio Almeida (Harvard University)",
            "Wagner Meira Jr. (UFMG)"
          ],
          "epfl_authors": [
            "Manoel Horta Ribeiro (EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M"
          ],
          "mdh_topics": [
            "Demonetisation",
            "Online activism",
            "Ad-revenue",
            "Twitter",
            "Brazil",
            "Counter-misinformation"
          ],
          "stage": "Mitigation",
          "relevance": 5,
          "about": "Study of the Sleeping Giants Brasil playbook, an online activism model that pressures companies to pull advertising from outlets spreading fake news and hate speech. Analyzing three campaigns with the group's tweets and the targeted companies using causal-inference methods, it finds the boycott requests worked at the company level: initial requests succeeded in 83.85 percent of cases, with most responses coming within a week. The campaigns produced no significant change in the targeted outlets' audience engagement or search interest over the following six months. Ad-revenue starvation succeeds in getting advertisers to leave, while reducing the outlets' actual reach does not.",
          "why": "It measures the real-world effectiveness of an advertising-boycott campaign aimed at outlets spreading fake news and hate speech.",
          "data": "Text (Portuguese Twitter + Google Trends); 1560 tweets from @slpng_giants_pt, 192 targeted companies, 3 SGB campaigns May-Sept 2020",
          "themes": [
            "Content moderation & enforcement",
            "Economics & incentives of MDH"
          ],
          "subtopics": [
            "Ad-revenue disruption",
            "Advertiser boycotts",
            "Consumer-led enforcement",
            "Demonetisation"
          ],
          "key_terms": [
            "Sleeping Giants",
            "Online activism",
            "Advertiser boycott",
            "Brazil",
            "Advertising boycott"
          ],
          "models": [
            "Perspective API",
            "SentiStrength"
          ],
          "method_qualifiers": [
            "Causal inference",
            "Sentiment analysis",
            "Synthetic control",
            "Toxicity classification"
          ],
          "events_cases": [
            "2020 Brazilian political climate",
            "COVID-19 pandemic",
            "Sleeping Giants Brasil campaigns (2020)"
          ],
          "built_at_epfl": [],
          "data_description": "1560 SGB tweets, 166 companies, about 1.1M mention tweets, 2020.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Twitter/X"
          ],
          "region_country": [
            "Brazil"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Demonetisation",
              "Consumer-led enforcement"
            ],
            "Economics & incentives of MDH": [
              "Ad-revenue disruption",
              "Advertiser boycotts"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Demonetisation"
            ],
            "Economics & incentives of MDH": [
              "Ad-revenue disruption"
            ]
          },
          "tentative": false
        },
        {
          "title": "Deplatforming Norm-Violating Influencers on Social Media Reduces Overall Online Attention Toward Them",
          "wid": "deplatforming-norm-violating-influencers-on-social-medi",
          "type": "publication",
          "year": 2025,
          "venue": "CSCW 2025",
          "link": "https://arxiv.org/abs/2401.01253",
          "authors": [
            "Manoel Horta Ribeiro (EPFL)",
            "Shagun Jhaver (Rutgers)",
            "Jordi Cluet i Martinell (EPFL)",
            "Marie Reignier-Tayar (EPFL)",
            "Robert West (EPFL)"
          ],
          "epfl_authors": [
            "Manoel Horta Ribeiro (EPFL)",
            "Jordi Cluet i Martinell (EPFL)",
            "Marie Reignier-Tayar (EPFL)",
            "Robert West (EPFL)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_topics": [
            "Deplatforming",
            "Content moderation",
            "Influencers",
            "Misinformation",
            "Online attention",
            "Causal inference"
          ],
          "stage": "Mitigation",
          "relevance": 4,
          "about": "Quasi-experimental study of what happens to a norm-violating influencer's overall online presence after removal from a major platform, addressing the worry that bans simply push attention elsewhere. Using a large dataset of deplatforming events, it tracks platform-agnostic attention through Google search interest and Wikipedia pageviews. Twelve months after deplatforming, attention had fallen by 63 percent on Google and 43 percent on Wikipedia, with people banned specifically for spreading misinformation reduced further than those banned for other reasons. Both permanent and temporary bans were effective.",
          "why": "It provides causal evidence that deplatforming reduces attention to norm-violating influencers, including those banned for misinformation.",
          "data": "Text/Web metrics; 165 deplatforming events / 101 influencers; Google Trends, Wikipedia pageviews, Media Cloud",
          "themes": [
            "Content moderation & enforcement"
          ],
          "subtopics": [
            "Ban effectiveness",
            "Deplatforming",
            "Temporary bans"
          ],
          "key_terms": [
            "Deplatforming",
            "Online attention",
            "Influencers",
            "Content moderation",
            "Difference-in-differences"
          ],
          "models": [],
          "method_qualifiers": [
            "Causal inference",
            "Difference-in-differences"
          ],
          "events_cases": [
            "Alex Jones ban"
          ],
          "built_at_epfl": [
            {
              "name": "Deplatforming events and online-attention dataset",
              "kind": "dataset"
            },
            {
              "name": "Deplatforming events dataset",
              "kind": "dataset",
              "url": "https://github.com/epfl-dlab/deplatforming_influencers",
              "evidence": "README of github.com/epfl-dlab/deplatforming_influencers: \"# Deplatforming Influencers / Code and Data for \\\"Deplatforming Norm-Violating Influencers on Social Media Reduces Overall Online Attention Toward Them.\\\"\" and, under **Data:**, \"1. `banned_unfiltered.csv`: Dataset of banned users (unfiltered). 2. `banned_final.csv`: Dataset of banned users (filtered). 3. `joined_final.csv`: Attention traces of banned users (linked to #2 above).\""
            }
          ],
          "data_description": "165 deplatforming events, 101 influencers, attention traces 2016-2021.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Facebook",
            "Instagram",
            "Reddit",
            "Twitter/X",
            "YouTube"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Content moderation & enforcement": [
              "Deplatforming",
              "Ban effectiveness",
              "Temporary bans"
            ]
          },
          "theme_qualifiers_canonical": {
            "Content moderation & enforcement": [
              "Ban effectiveness",
              "Deplatforming"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "emanuela-boros",
      "name": "Emanuela Boros",
      "unit": "L3i, La Rochelle Université",
      "faculty": "formerly DHLAB, EPFL (until January 2026)",
      "mdh_focus": [],
      "dataTypes": [],
      "techTypes": [],
      "stage": "Prevention + Monitoring",
      "publications": [
        {
          "title": "Linguistic Mechanisms of Cultural Heritage Weaponization in Digital Encyclopedic Discourse: Evidence from Wikipedia Revision Corpora",
          "wid": "linguistic-mechanisms-of-cultural-heritage-weaponizatio",
          "type": "publication",
          "year": 2026,
          "venue": "Scientific Journal of Dragomanov Ukrainian State University, Series 9, Issue 31",
          "link": "https://doi.org/10.31392/udu-nc.series9.2026.31.09",
          "authors": [
            "Hamest S. Tamrazyan",
            "Maxime Garambois",
            "Emanuela Boros"
          ],
          "epfl_authors": [
            "Hamest S. Tamrazyan",
            "Maxime Garambois",
            "Emanuela Boros"
          ],
          "about": "Article identifying six recurrent strategies of discursive manipulation in approximately 200000 Wikipedia revisions related to the Russia-Ukraine conflict, including agent-downgrading through derivational prefixation, spelling changes to place names such as Kyiv to Kiev, register substitution along the euphemism-dysphemism axis and the erasure of citations. Revisions from 351 articles were classified by GPT-4 as weaponized or not, and a subset of about 7000 was reviewed by two expert annotators, with agreement of 0.73 (Cohen's κ). The study finds that cultural weaponization on open encyclopedic platforms operates not through explicit falsehood but through accumulated micro-level edits that individually comply with editorial norms, so standard fact-checking tools are insufficient.",
          "themes": [
            "Media framing & narrative analysis"
          ],
          "subtopics": [
            "Euphemism-dysphemism substitution",
            "Issue framing",
            "Narrative shift over time"
          ],
          "mdh_focus": [
            "D"
          ],
          "mdh_relevance": "direct",
          "relevance": 5,
          "stage": "Monitoring",
          "key_terms": [
            "Cultural heritage weaponization",
            "Russia-Ukraine conflict",
            "Epistemic injustice",
            "Critical Discourse Analysis",
            "Lexical framing"
          ],
          "models": [
            "GPT-4",
            "Sentence-BERT"
          ],
          "method_qualifiers": [
            "Critical Discourse Analysis",
            "LLM-as-a-judge",
            "Manual annotation",
            "Zero-shot classification"
          ],
          "events_cases": [
            "Russia-Ukraine conflict",
            "Russo-Ukrainian war"
          ],
          "built_at_epfl": [],
          "data_description": "Approx. 200000 Wikipedia revisions from 351 Russia-Ukraine conflict articles.",
          "platform": [
            "Wikipedia"
          ],
          "region_country": [
            "Russia",
            "Ukraine"
          ],
          "targeted_group": [
            "Ukrainians"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Media framing & narrative analysis": [
              "Narrative shift over time",
              "Euphemism-dysphemism substitution",
              "Issue framing"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media framing & narrative analysis": [
              "Issue framing",
              "Narrative shift over time"
            ]
          },
          "tentative": false
        },
        {
          "title": "Apertus: Democratizing Open and Compliant LLMs for Global Language Environments",
          "wid": "apertus-democratizing-open-and-compliant-llms-for-globa",
          "type": "publication",
          "year": 2025,
          "venue": "arXiv 2509.14233",
          "link": "https://arxiv.org/abs/2509.14233",
          "authors": [
            "Alejandro Hernández-Cano",
            "Alexander Hägele",
            "Allen Hao Huang",
            "Angelika Romanou",
            "Antoni-Joan Solergibert",
            "Barna Pasztor",
            "Bettina Messmer",
            "Dhia Garbaya",
            "Eduard Frank Ďurech",
            "Ido Hakimi",
            "Juan García Giraldo",
            "Mete Ismayilzada",
            "Negar Foroutan",
            "Skander Moalla",
            "Tiancheng Chen",
            "Vinko Sabolčec",
            "Yixuan Xu",
            "Michael Aerni",
            "Badr AlKhamissi",
            "Inés Altemir Mariñas",
            "Mohammad Hossein Amani",
            "Matin Ansaripour",
            "Ilia Badanin",
            "Harold Benoit",
            "Emanuela Boros",
            "Nicholas Browning",
            "Fabian Bösch",
            "Maximilian Böther",
            "Niklas Canova",
            "Camille Challier",
            "Clement Charmillot",
            "Jonathan Coles",
            "Jan Deriu",
            "Arnout Devos",
            "Lukas Drescher",
            "Daniil Dzenhaliou",
            "Maud Ehrmann",
            "Dongyang Fan",
            "Simin Fan",
            "Silin Gao",
            "Miguel Gila",
            "María Grandury",
            "Diba Hashemi",
            "Alexander Hoyle",
            "Jiaming Jiang",
            "Mark Klein",
            "Andrei Kucharavy",
            "Anastasiia Kucherenko",
            "Frederike Lübeck",
            "Roman Machacek",
            "Theofilos Manitaras",
            "Andreas Marfurt",
            "Kyle Matoba",
            "Simon Matrenok",
            "Henrique Mendonça",
            "Fawzi Roberto Mohamed",
            "Syrielle Montariol",
            "Luca Mouchel",
            "Sven Najem-Meyer",
            "Jingwei Ni",
            "Gennaro Oliva",
            "Matteo Pagliardini",
            "Elia Palme",
            "Andrei Panferov",
            "Léo Paoletti",
            "Marco Passerini",
            "Ivan Pavlov",
            "Auguste Poiroux",
            "Kaustubh Ponkshe",
            "Nathan Ranchin",
            "Javi Rando",
            "Mathieu Sauser",
            "Jakhongir Saydaliev",
            "Muhammad Ali Sayfiddinov",
            "Marian Schneider",
            "Stefano Schuppli",
            "Marco Scialanga",
            "Andrei Semenov",
            "Kumar Shridhar",
            "Raghav Singhal",
            "Anna Sotnikova",
            "Alexander Sternfeld",
            "Ayush Kumar Tarun",
            "Paul Teiletche",
            "Jannis Vamvas",
            "Xiaozhe Yao",
            "Hao Zhao",
            "Alexander Ilic",
            "Ana Klimovic",
            "Andreas Krause",
            "Caglar Gulcehre",
            "David Rosenthal",
            "Elliott Ash",
            "Florian Tramèr",
            "Joost VandeVondele",
            "Livio Veraldi",
            "Martin Rajman",
            "Thomas Schulthess",
            "Torsten Hoefler",
            "Antoine Bosselut",
            "Martin Jaggi",
            "Imanol Schlag"
          ],
          "epfl_authors": [
            "Alejandro Hernández-Cano",
            "Alexander Hägele",
            "Allen Hao Huang",
            "Angelika Romanou",
            "Antoni-Joan Solergibert",
            "Bettina Messmer",
            "Dhia Garbaya",
            "Eduard Frank Ďurech",
            "Juan García Giraldo",
            "Mete Ismayilzada",
            "Negar Foroutan",
            "Skander Moalla",
            "Vinko Sabolčec",
            "Yixuan Xu",
            "Badr AlKhamissi",
            "Inés Altemir Mariñas",
            "Mohammad Hossein Amani",
            "Matin Ansaripour",
            "Ilia Badanin",
            "Harold Benoit",
            "Emanuela Boros",
            "Camille Challier",
            "Clement Charmillot",
            "Daniil Dzenhaliou",
            "Maud Ehrmann",
            "Dongyang Fan",
            "Simin Fan",
            "Silin Gao",
            "María Grandury",
            "Diba Hashemi",
            "Jiaming Jiang",
            "Kyle Matoba",
            "Simon Matrenok",
            "Syrielle Montariol",
            "Luca Mouchel",
            "Sven Najem-Meyer",
            "Matteo Pagliardini",
            "Léo Paoletti",
            "Ivan Pavlov",
            "Auguste Poiroux",
            "Kaustubh Ponkshe",
            "Nathan Ranchin",
            "Mathieu Sauser",
            "Jakhongir Saydaliev",
            "Marco Scialanga",
            "Andrei Semenov",
            "Raghav Singhal",
            "Anna Sotnikova",
            "Ayush Kumar Tarun",
            "Paul Teiletche",
            "Hao Zhao",
            "Caglar Gulcehre",
            "Martin Rajman",
            "Antoine Bosselut",
            "Martin Jaggi"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_topics": [
            "Open-source LLM",
            "Data compliance",
            "Multilingual",
            "Toxic-content filtering",
            "Swiss AI Initiative",
            "Digital sovereignty"
          ],
          "stage": "Prevention + Monitoring",
          "relevance": 3,
          "about": "Technical report for Apertus, a fully open suite of large language models trained on web-scale multilingual data spanning more than 1800 languages, with around 40 percent of pretraining data non-English. It trains only on openly available data, respects robots.txt exclusions, and filters out non-permissive, toxic and personally identifiable content, with everything released under a permissive license so the pipeline can be audited. Released at 8B and 70B scales and trained on 15 trillion tokens, it approaches state-of-the-art results among fully open models on multilingual benchmarks while training only on compliant data.",
          "why": "Its toxic-content filtering and auditable, opt-out-respecting pretraining make it a trustworthy base model for downstream MDH work.",
          "data": "Text, 15 trillion tokens across 1800+ languages",
          "lab": [
            "NLP",
            "MLO",
            "CLAIRE"
          ],
          "themes": [
            "AI safety",
            "Platform governance & regulation",
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Constitutional AI",
            "Implicit hate speech",
            "Misuse risk assessment",
            "Model-generated toxicity",
            "Safety alignment"
          ],
          "key_terms": [
            "Data compliance",
            "Swiss AI Charter",
            "Constitutional AI",
            "Constitutional AI alignment",
            "EU AI Act compliance"
          ],
          "models": [
            "Apertus-70B-Instruct"
          ],
          "method_qualifiers": [
            "Goldfish loss",
            "Constitutional AI",
            "Preference optimisation",
            "QRPO alignment"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Apertus-70B-Instruct",
              "kind": "model"
            },
            {
              "name": "Apertus-sft-mixture",
              "kind": "dataset"
            },
            {
              "name": "Swiss AI Charter",
              "kind": "framework",
              "url": "https://www.apertus-ai.org/pages/charter/",
              "evidence": "The Apertus Charter defines alignment principles for AI systems developed under the Swiss AI Initiative, rooted in Switzerland's constitutional values and democratic traditions. (Previously referred to as the Swiss AI Charter) Version 1.0 - August 2025"
            },
            {
              "name": "SwitzerlandQA",
              "kind": "benchmark",
              "evidence": "Paper Appendix K: \"we test the model on its understanding of Switzerland's environment by developing a novel benchmark SwitzerlandQA specifically tailored to Switzerland's context. [...] The benchmark represents 26 cantons, with each canton having at least 200 questions, yielding 9167 unique items per language across domains and levels of granularity.\" The artefact list names it as swiss-ai/switzerland_qa, but no public URL was found - see notes."
            },
            {
              "name": "apertus-pretrain-toxicity",
              "kind": "model",
              "url": "https://huggingface.co/swiss-ai/apertus-pretrain-toxicity",
              "evidence": "Model card: \"Language specific toxicity classifiers in English, French, German, Italian, Spanish, Portuguese, Polish, Chinese and Dutch, trained on PleIAs/ToxicCommons and SWSR-SexComments datasets.\" and \"The classifier checkpoints with the best accuracy on the held-out validation set are further employed to annotate the toxicity scores on FineWeb-2 and FineWeb.\" The paper cites this exact URL as footnote 13 in the toxicity-filtering section."
            },
            {
              "name": "RealToxicityPrompts-Llama-Subsampled",
              "kind": "benchmark",
              "url": "https://huggingface.co/datasets/swiss-ai/realtoxicityprompts",
              "evidence": "Paper: \"To integrate it in our benchmark harness, we sub-sample it to 10% of its size and switch the toxicity classifier model to Llama-Guard-3-8B (Fedorov et al., 2024) to allow fully-contained execution. We release this subsample, [48] as well as the LLaMA-Guard-3-8B implementation. [49] The resulting benchmark, RealToxicityPromptsLlama-Subsampled...\" with footnote 48 = https://huggingface.co/datasets/swiss-ai/realtoxicityprompts/tree/main/realtoxicityprompts_small. The dataset page confirms two subsets, realtoxicityprompts_full (99.4k rows) and realtoxicityprompts_small (10k rows)."
            }
          ],
          "follow_up": [
            {
              "what": "Swisscom deployed Apertus on its sovereign Swiss AI Platform for business customers on the day of release",
              "kind": "deployment",
              "url": "https://www.swisscom.ch/en/about/news/2025/09/02-apertus.html",
              "evidence": "Swisscom is proud to be among the first to deploy this pioneering large language model on our sovereign Swiss AI Platform. [...] As of today, Swisscom business customers will be able to access the Apertus model via Swisscom's sovereign Swiss AI platform."
            },
            {
              "what": "The Public AI Inference Utility became the official international deployer of Apertus, serving it worldwide",
              "kind": "deployment",
              "url": "https://publicai.co/stories/apertus",
              "evidence": "Public AI is proud to be the official international deployer for Apertus. To support Apertus, we've allocated over 115000 GPU-hours spread across 20 clusters in 5+ countries - just for the month of September."
            },
            {
              "what": "Apertus v1.1, a family of distilled models trained from the Apertus-8B-2509 teacher described in this report",
              "kind": "model",
              "url": "https://huggingface.co/swiss-ai/Apertus-v1.1-4B-Instruct",
              "evidence": "For more details refer to the original Apertus [technical report](https://arxiv.org/abs/2509.14233) and the new Apertus [distillation technical report](https://arxiv.org/abs/2605.29128)."
            },
            {
              "what": "Apertus v1.5, a multimodal continuation of the models released in this report",
              "kind": "successor-work",
              "url": "https://huggingface.co/swiss-ai/Apertus-v1.5-70B",
              "evidence": "The released models are the result of continued pretraining of Apertus 1.0, adding a multimodal mix of 4T tokens to the 8B model and 2T tokens to the 70B model."
            }
          ],
          "data_description": "15T tokens from 1800+ languages, filtered for toxicity and compliance.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "Switzerland"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Safety alignment",
              "Misuse risk assessment",
              "Constitutional AI"
            ],
            "Toxicity & harassment": [
              "Model-generated toxicity",
              "Implicit hate speech"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Misuse risk assessment",
              "Safety alignment"
            ],
            "Toxicity & harassment": [
              "Implicit hate speech",
              "Model-generated toxicity"
            ]
          },
          "tentative": false
        },
        {
          "title": "Narrative wars: Detecting the weaponisation of cultural heritage in the digital sphere",
          "wid": "cross-narrative-wars-project",
          "type": "project",
          "year": 2026,
          "venue": "CROSS 2026 (EPFL-UNIL Collaborative Research on Science and Society). Led by Hamest Tamrazyan (CDH, DHI-GE) and Emanuela Boros (CDH, DHI, DHLAB) at EPFL with Stephanie Prezioso and Hanna Perekhonda (FSSP, IEP) at UNIL. Emanuela Boros left EPFL in January 2026.",
          "link": "https://www.epfl.ch/schools/cdh/cross-2026/",
          "authors": [
            "Hamest Tamrazyan (EPFL DHI)",
            "Emanuela Boros (EPFL DHLab)",
            "Stéphanie Prezioso (UNIL IEP)",
            "Hanna Perekhonda (UNIL IEP)"
          ],
          "epfl_authors": [
            "Hamest Tamrazyan (EPFL DHI)",
            "Emanuela Boros (EPFL DHLab)",
            "Stéphanie Prezioso (UNIL IEP)",
            "Hanna Perekhonda (UNIL IEP)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "D"
          ],
          "mdh_topics": [
            "Narrative manipulation",
            "Wikipedia",
            "Conflict zones",
            "Language weaponisation",
            "Ukraine",
            "Nagorno-Karabakh"
          ],
          "stage": "Monitoring",
          "relevance": 5,
          "about": "Multi-year DHLAB project studying how language is weaponised in geopolitical conflict zones by tracking edits to Wikipedia across the Ukrainian, Armenian, Azerbaijani and English editions, with a focus on Ukraine and Nagorno-Karabakh. The work concentrates on subtle word-level shifts that re-anchor historical narratives over time rather than overt fake news, and the analysis is designed to stay neutral about which side is editing.",
          "why": "It studies covert narrative manipulation and incremental rewriting of contested history through coordinated Wikipedia edits in conflict zones.",
          "data": "Text (hundreds of thousands of Wikipedia edits in Ukrainian, Armenian, Azerbaijani, English language editions)",
          "themes": [
            "Media framing & narrative analysis"
          ],
          "subtopics": [
            "Cultural heritage narratives",
            "Issue framing",
            "Narrative reframing",
            "Wikipedia editing manipulation"
          ],
          "key_terms": [
            "Cultural heritage weaponisation",
            "Wikipedia",
            "Armenia",
            "Ukraine",
            "Conflict narratives"
          ],
          "models": [],
          "method_qualifiers": [
            "Thematic analysis",
            "Historical interpretation"
          ],
          "events_cases": [
            "Ukraine"
          ],
          "built_at_epfl": [],
          "data_description": "Wikipedia articles and edits concerning Armenia and Ukraine.",
          "follow_up_checked": "2026-09-02",
          "platform": [
            "Wikipedia"
          ],
          "region_country": [
            "Armenia",
            "Ukraine"
          ],
          "targeted_group": [
            "Armenians",
            "Ukrainians"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Influence operations & coordinated manipulation": [
              "Narrative reframing",
              "Wikipedia editing manipulation"
            ],
            "Media framing & narrative analysis": [
              "Narrative shift over time",
              "Issue framing",
              "Cultural heritage narratives"
            ]
          },
          "theme_qualifiers_canonical": {
            "Influence operations & coordinated manipulation": [
              "Narrative reframing"
            ],
            "Media framing & narrative analysis": [
              "Issue framing",
              "Narrative shift over time"
            ]
          },
          "tentative": true
        },
        {
          "title": "Impresso project",
          "wid": "impresso-project",
          "type": "project",
          "year": "ongoing",
          "venue": "SNSF-funded research programme",
          "link": "https://impresso-project.ch/",
          "authors": [
            "Maud Ehrmann (EPFL DHLab)",
            "Simon Clematide (UZH)",
            "Raphaelle Ruppen Coutaz (UNIL)",
            "Marten During (Uni Luxembourg C2DH)",
            "Emanuela Boros (EPFL DHLab)",
            "Pauline Conti (EPFL DHLab)",
            "Marina Butyrskaya Moyer (EPFL DHLab)",
            "Arthur Michelet (UNIL)",
            "Martin Grandjean (EPFL DHLab)",
            "Caio Mello (Uni Luxembourg C2DH)",
            "Cao Vy (Uni Luxembourg C2DH)",
            "Daniele Guido (Uni Luxembourg C2DH)",
            "Estelle Bunout (Uni Luxembourg C2DH)",
            "Ferdaous Affan (Uni Luxembourg C2DH)",
            "Kirill Mitsurov (Uni Luxembourg C2DH)",
            "Roman Kalyakin (Uni Luxembourg C2DH)",
            "Andrianos Michail (UZH)",
            "Juri Opitz (UZH)",
            "Kaspar Beelen (SAS, University of London)"
          ],
          "epfl_authors": [
            "Maud Ehrmann (EPFL DHLab)",
            "Emanuela Boros (EPFL DHLab)",
            "Pauline Conti (EPFL DHLab)",
            "Marina Butyrskaya Moyer (EPFL DHLab)",
            "Raphaelle Ruppen Coutaz (UNIL)",
            "Arthur Michelet (UNIL)",
            "Martin Grandjean (EPFL DHLab)"
          ],
          "mdh_relevance": "infrastructure",
          "mdh_focus": [
            "M",
            "D"
          ],
          "mdh_topics": [
            "Historical newspapers",
            "Knowledge infrastructure",
            "Sourced corpora"
          ],
          "stage": "NA",
          "relevance": 2,
          "about": "Interdisciplinary research programme applying machine learning to historical media, semantically enriching and connecting digitised newspapers and radio broadcasts so researchers can explore them across languages, countries and decades. It produces public web applications, datasets and models. On the press side it complements Time Machine as historical knowledge infrastructure, building a stable, sourced record of past media.",
          "why": "It builds a stable, sourced corpus of historical press and broadcast media, an indirect counterweight to a less anchored information environment.",
          "data": "Historical newspaper corpora (Swiss-French / Luxembourgish / multilingual press)",
          "themes": [
            "NA"
          ],
          "subtopics": [],
          "key_terms": [
            "Radio archives",
            "Digital humanities",
            "Historical media",
            "NLP",
            "Europe"
          ],
          "models": [],
          "method_qualifiers": [
            "Semantic enrichment"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Impresso",
              "kind": "platform",
              "url": "https://impresso-project.ch/app/",
              "evidence": "https://impresso-project.ch/the-app/ : \"The Impresso web app (https://impresso-project.ch/app/) offers a graphical user interface for exploration and the compilation of research datasets. It builds on the application developed during the first Impresso project. As part of the second project, the web app is revised to facilitate access to different types of historical media, such as audio recordings, and various types of text, e.g., typescripts, transcribed speech, or radio programming schedules.\" The first project is confirmed as Kaplan's at https://www.epfl.ch/labs/dhlab/projects/impresso-old/ : \"impresso - Media monitoring of the past. Mining 200 years of historical newspapers\", PI Frederic Kaplan (DHLAB, EPFL), September 2017 - August 2020, SNSF Sinergia."
            },
            {
              "name": "Impresso corpus",
              "kind": "dataset"
            },
            {
              "name": "Impresso Data Lab",
              "kind": "tool"
            },
            {
              "name": "Impresso web application",
              "kind": "tool"
            }
          ],
          "follow_up": [
            {
              "what": "Impresso II - 'Media Monitoring of the Past II. Beyond Borders: Connecting Historical Newspapers and Radio', the funded continuation (SNSF 213585 Sinergia + FNR 17498891 INTER, September 2023 to February 2027)",
              "kind": "project",
              "url": "https://www.epfl.ch/labs/dhlab/impresso-media-monitoring-of-the-past-ii-beyond-borders-connecting-historical-newspapers-and-radio/",
              "evidence": "\"funded by the Swiss National Science Foundation (SNSF 213585) and the Luxembourg National Research Fund (FNR 17498891) as part of the Sinergia / INTER funding programmes from September 2023 until February 2027.\" Applicant institutions: DHLAB at EPFL, University of Lausanne History Department, University of Zurich Institute of Computational Linguistics, University of Luxembourg C2DH. That it continues the first project is stated at https://impresso-project.ch/the-app/ : \"It builds on the application developed during the first Impresso project. As part of the second project, the web app is revised...\""
            },
            {
              "what": "HIPE (Identifying Historical People, Places and other Entities) - a series of evaluation campaigns / shared tasks on named entity recognition and linking in multilingual historical documents, spun out of the project",
              "kind": "benchmark",
              "url": "https://impresso-project.ch/hipe/",
              "evidence": "\"Initiated during the first Impresso project, HIPE (Identifying Historical People, Places and other Entities) is a series of evaluation campaigns, or shared tasks, on named entity recognition and linking in multilingual historical documents.\" and \"The HIPE-eval initiative will be continued during the second Impresso project and will gradually evolve to include further information extraction tasks on historical documents.\""
            },
            {
              "what": "Impresso data lab - API-based data access and annotation services with executable Jupyter notebooks, built in the second project",
              "kind": "deployment",
              "url": "https://impresso-project.ch/datalab",
              "evidence": "https://impresso-project.ch/the-app/ , section '2. Impresso data lab': \"The forthcoming Impresso data lab is an infrastructure for data access and annotation services via APIs, along with their integration in executable Jupyter notebooks. The data lab will provide researchers with examples based on experiments with data-driven analyses of the Impresso corpus. Annotation services offer access to semantic indexing models and the ability to relate external data to the corpus.\""
            },
            {
              "what": "Impresso model and dataset releases on HuggingFace: 27 models (historical NER, named entity linking, topic inference, OCR quality assessment, ad classification, multilingual BERT variants trained on historical media) and 10 datasets (including the HIPE NER and NEL sets)",
              "kind": "model",
              "url": "https://huggingface.co/impresso-project",
              "evidence": "Organisation description: \"an interdisciplinary research project using machine learning to transform how historical media are processed, enriched, explored, and studied across modalities\", developing a web application and datalab giving access to a multilingual corpus of historical newspapers and radio broadcasts; funded by the Swiss National Science Foundation."
            },
            {
              "what": "Impresso open-source codebase on GitHub (impresso-frontend, impresso-text-acquisition, impresso-pipelines, impresso-datalab-notebooks, CLEF-HIPE-2020, impresso-schemas and others)",
              "kind": "repository",
              "url": "https://github.com/impresso",
              "evidence": "Organisation description: \"Impresso - Media Monitoring of the Past ... an interdisciplinary research project that uses machine learning to pursue a paradigm shift in the processing, semantic enrichment, representation, exploration and study of historical media across modalities, temporal, linguistic, and national borders.\" The org states funding for two periods, 2017-2020 and 2023-2027, and that \"It develops the Impresso Web App and the upcoming Impresso Datalab.\""
            }
          ],
          "data_description": "Corpus of European newspaper and radio archives, many countries and languages.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "Europe"
          ],
          "label_version": "v5",
          "theme_qualifiers": {},
          "theme_qualifiers_canonical": {},
          "tentative": true
        }
      ]
    },
    {
      "id": "syrielle-montariol",
      "name": "Syrielle Montariol",
      "unit": "ISIR, Sorbonne Université (CNRS)",
      "faculty": "formerly NLP Lab, EPFL (postdoc 2023 to 2025)",
      "mdh_focus": [],
      "dataTypes": [],
      "techTypes": [],
      "stage": "Prevention + Monitoring",
      "publications": [
        {
          "title": "Multilingual Auxiliary Tasks Training: Bridging the Gap between Languages for Zero-Shot Transfer of Hate Speech Detection Models",
          "wid": "multilingual-auxiliary-tasks-training-bridging-the-gap-",
          "type": "publication",
          "year": 2022,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2210.13029",
          "authors": [
            "Syrielle Montariol",
            "Arij Riabi",
            "Djamé Seddah"
          ],
          "about": "Written at INRIA Paris, before Syrielle Montariol joined EPFL. Study of zero-shot cross-lingual transfer of hate speech detection models across English, Italian and Spanish datasets on hate speech against women and immigrants, rebuilt as strictly comparable corpora. It trains XLM-R on hate speech in the source language jointly with auxiliary tasks in all three languages: sentiment analysis, named entity recognition (NER) and syntactic Universal Dependency tasks. Sentiment analysis and NER improve transfer for the immigrants domain, by 2.5 percentage points combined, but for the women domain they help almost only in the monolingual setting, while syntactic tasks cause a large drop in most cases. On the HateCheck test suite, slurs show the largest loss under zero-shot transfer.",
          "themes": [
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Cross-lingual transfer",
            "Identity-targeted hate",
            "Online harassment of women"
          ],
          "mdh_focus": [
            "H"
          ],
          "mdh_relevance": "direct",
          "relevance": 5,
          "stage": "Monitoring",
          "key_terms": [
            "Zero-shot cross-lingual transfer",
            "Auxiliary task training",
            "Hate speech detection",
            "Anti-immigrant hate speech",
            "Auxiliary tasks"
          ],
          "models": [
            "XLM-R",
            "XLM-T",
            "mBERT"
          ],
          "method_qualifiers": [
            "Multi-task learning",
            "Cross-lingual transfer",
            "Auxiliary task training"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Comparable multilingual hate speech corpora",
              "kind": "dataset"
            }
          ],
          "data_description": "Comparable English/Spanish/Italian tweet corpora for hate speech plus auxiliary text datasets.",
          "platform": [
            "Facebook",
            "Twitter"
          ],
          "targeted_group": [
            "Immigrants",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Toxicity & harassment": [
              "Cross-lingual transfer",
              "Identity-targeted hate",
              "Online harassment of women"
            ]
          },
          "theme_qualifiers_canonical": {
            "Toxicity & harassment": [
              "Identity-targeted hate",
              "Online harassment of women"
            ]
          },
          "tentative": false
        },
        {
          "title": "Tâches Auxiliaires Multilingues pour le Transfert de Modèles de Détection de Discours Haineux",
          "wid": "t-ches-auxiliaires-multilingues-pour-le-transfert-de-mo",
          "type": "publication",
          "year": 2022,
          "venue": "TALN 2022",
          "link": "https://aclanthology.org/2022.jeptalnrecital-taln.41/",
          "authors": [
            "Arij Riabi",
            "Syrielle Montariol",
            "Djamé Seddah"
          ],
          "about": "Written at INRIA Paris, before Syrielle Montariol joined EPFL. Paper on zero-shot cross-lingual transfer of hate speech detection models, which cultural variation in hateful content across languages often makes difficult. Using tweets annotated for hate against women and against immigrants in English, Italian and Spanish, it fine-tunes XLM-R jointly with auxiliary tasks using data from all three languages: sentiment analysis, named entity recognition (NER) and syntactic Universal Dependencies (UD) tasks. Sentiment analysis and NER improve cross-lingual transfer for hate against immigrants, while UD tasks degrade it. For hate against women, where zero-shot transfer to the Spanish and Italian test sets fails, auxiliary NER training and a language model trained on tweets (XLM-T) are the two most effective solutions.",
          "themes": [
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Cultural gap",
            "Identity-targeted hate",
            "Online harassment of women"
          ],
          "mdh_focus": [
            "H"
          ],
          "mdh_relevance": "direct",
          "relevance": 5,
          "stage": "Monitoring",
          "key_terms": [
            "Hate speech detection",
            "Multilingual auxiliary tasks",
            "Cultural gap",
            "XLM-R",
            "Anti-immigrant hate speech"
          ],
          "models": [
            "XLM-R",
            "XLM-T"
          ],
          "method_qualifiers": [
            "Multi-task learning",
            "Zero-shot transfer",
            "Cross-lingual comparison",
            "Zero-shot classification"
          ],
          "events_cases": [],
          "built_at_epfl": [],
          "data_description": "2591 tweets per corpus, 3 languages, 2 domains (women, immigrants).",
          "platform": [
            "Twitter/X"
          ],
          "targeted_group": [
            "Immigrants",
            "Migrants",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Toxicity & harassment": [
              "Identity-targeted hate",
              "Online harassment of women",
              "Cultural gap"
            ]
          },
          "theme_qualifiers_canonical": {
            "Toxicity & harassment": [
              "Identity-targeted hate",
              "Online harassment of women"
            ]
          },
          "tentative": false
        },
        {
          "title": "Apertus: Democratizing Open and Compliant LLMs for Global Language Environments",
          "wid": "apertus-democratizing-open-and-compliant-llms-for-globa",
          "type": "publication",
          "year": 2025,
          "venue": "arXiv 2509.14233",
          "link": "https://arxiv.org/abs/2509.14233",
          "authors": [
            "Alejandro Hernández-Cano",
            "Alexander Hägele",
            "Allen Hao Huang",
            "Angelika Romanou",
            "Antoni-Joan Solergibert",
            "Barna Pasztor",
            "Bettina Messmer",
            "Dhia Garbaya",
            "Eduard Frank Ďurech",
            "Ido Hakimi",
            "Juan García Giraldo",
            "Mete Ismayilzada",
            "Negar Foroutan",
            "Skander Moalla",
            "Tiancheng Chen",
            "Vinko Sabolčec",
            "Yixuan Xu",
            "Michael Aerni",
            "Badr AlKhamissi",
            "Inés Altemir Mariñas",
            "Mohammad Hossein Amani",
            "Matin Ansaripour",
            "Ilia Badanin",
            "Harold Benoit",
            "Emanuela Boros",
            "Nicholas Browning",
            "Fabian Bösch",
            "Maximilian Böther",
            "Niklas Canova",
            "Camille Challier",
            "Clement Charmillot",
            "Jonathan Coles",
            "Jan Deriu",
            "Arnout Devos",
            "Lukas Drescher",
            "Daniil Dzenhaliou",
            "Maud Ehrmann",
            "Dongyang Fan",
            "Simin Fan",
            "Silin Gao",
            "Miguel Gila",
            "María Grandury",
            "Diba Hashemi",
            "Alexander Hoyle",
            "Jiaming Jiang",
            "Mark Klein",
            "Andrei Kucharavy",
            "Anastasiia Kucherenko",
            "Frederike Lübeck",
            "Roman Machacek",
            "Theofilos Manitaras",
            "Andreas Marfurt",
            "Kyle Matoba",
            "Simon Matrenok",
            "Henrique Mendonça",
            "Fawzi Roberto Mohamed",
            "Syrielle Montariol",
            "Luca Mouchel",
            "Sven Najem-Meyer",
            "Jingwei Ni",
            "Gennaro Oliva",
            "Matteo Pagliardini",
            "Elia Palme",
            "Andrei Panferov",
            "Léo Paoletti",
            "Marco Passerini",
            "Ivan Pavlov",
            "Auguste Poiroux",
            "Kaustubh Ponkshe",
            "Nathan Ranchin",
            "Javi Rando",
            "Mathieu Sauser",
            "Jakhongir Saydaliev",
            "Muhammad Ali Sayfiddinov",
            "Marian Schneider",
            "Stefano Schuppli",
            "Marco Scialanga",
            "Andrei Semenov",
            "Kumar Shridhar",
            "Raghav Singhal",
            "Anna Sotnikova",
            "Alexander Sternfeld",
            "Ayush Kumar Tarun",
            "Paul Teiletche",
            "Jannis Vamvas",
            "Xiaozhe Yao",
            "Hao Zhao",
            "Alexander Ilic",
            "Ana Klimovic",
            "Andreas Krause",
            "Caglar Gulcehre",
            "David Rosenthal",
            "Elliott Ash",
            "Florian Tramèr",
            "Joost VandeVondele",
            "Livio Veraldi",
            "Martin Rajman",
            "Thomas Schulthess",
            "Torsten Hoefler",
            "Antoine Bosselut",
            "Martin Jaggi",
            "Imanol Schlag"
          ],
          "epfl_authors": [
            "Alejandro Hernández-Cano",
            "Alexander Hägele",
            "Allen Hao Huang",
            "Angelika Romanou",
            "Antoni-Joan Solergibert",
            "Bettina Messmer",
            "Dhia Garbaya",
            "Eduard Frank Ďurech",
            "Juan García Giraldo",
            "Mete Ismayilzada",
            "Negar Foroutan",
            "Skander Moalla",
            "Vinko Sabolčec",
            "Yixuan Xu",
            "Badr AlKhamissi",
            "Inés Altemir Mariñas",
            "Mohammad Hossein Amani",
            "Matin Ansaripour",
            "Ilia Badanin",
            "Harold Benoit",
            "Emanuela Boros",
            "Camille Challier",
            "Clement Charmillot",
            "Daniil Dzenhaliou",
            "Maud Ehrmann",
            "Dongyang Fan",
            "Simin Fan",
            "Silin Gao",
            "María Grandury",
            "Diba Hashemi",
            "Jiaming Jiang",
            "Kyle Matoba",
            "Simon Matrenok",
            "Syrielle Montariol",
            "Luca Mouchel",
            "Sven Najem-Meyer",
            "Matteo Pagliardini",
            "Léo Paoletti",
            "Ivan Pavlov",
            "Auguste Poiroux",
            "Kaustubh Ponkshe",
            "Nathan Ranchin",
            "Mathieu Sauser",
            "Jakhongir Saydaliev",
            "Marco Scialanga",
            "Andrei Semenov",
            "Raghav Singhal",
            "Anna Sotnikova",
            "Ayush Kumar Tarun",
            "Paul Teiletche",
            "Hao Zhao",
            "Caglar Gulcehre",
            "Martin Rajman",
            "Antoine Bosselut",
            "Martin Jaggi"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_topics": [
            "Open-source LLM",
            "Data compliance",
            "Multilingual",
            "Toxic-content filtering",
            "Swiss AI Initiative",
            "Digital sovereignty"
          ],
          "stage": "Prevention + Monitoring",
          "relevance": 3,
          "about": "Technical report for Apertus, a fully open suite of large language models trained on web-scale multilingual data spanning more than 1800 languages, with around 40 percent of pretraining data non-English. It trains only on openly available data, respects robots.txt exclusions, and filters out non-permissive, toxic and personally identifiable content, with everything released under a permissive license so the pipeline can be audited. Released at 8B and 70B scales and trained on 15 trillion tokens, it approaches state-of-the-art results among fully open models on multilingual benchmarks while training only on compliant data.",
          "why": "Its toxic-content filtering and auditable, opt-out-respecting pretraining make it a trustworthy base model for downstream MDH work.",
          "data": "Text, 15 trillion tokens across 1800+ languages",
          "lab": [
            "NLP",
            "MLO",
            "CLAIRE"
          ],
          "themes": [
            "AI safety",
            "Platform governance & regulation",
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Constitutional AI",
            "Implicit hate speech",
            "Misuse risk assessment",
            "Model-generated toxicity",
            "Safety alignment"
          ],
          "key_terms": [
            "Data compliance",
            "Swiss AI Charter",
            "Constitutional AI",
            "Constitutional AI alignment",
            "EU AI Act compliance"
          ],
          "models": [
            "Apertus-70B-Instruct"
          ],
          "method_qualifiers": [
            "Goldfish loss",
            "Constitutional AI",
            "Preference optimisation",
            "QRPO alignment"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Apertus-70B-Instruct",
              "kind": "model"
            },
            {
              "name": "Apertus-sft-mixture",
              "kind": "dataset"
            },
            {
              "name": "Swiss AI Charter",
              "kind": "framework",
              "url": "https://www.apertus-ai.org/pages/charter/",
              "evidence": "The Apertus Charter defines alignment principles for AI systems developed under the Swiss AI Initiative, rooted in Switzerland's constitutional values and democratic traditions. (Previously referred to as the Swiss AI Charter) Version 1.0 - August 2025"
            },
            {
              "name": "SwitzerlandQA",
              "kind": "benchmark",
              "evidence": "Paper Appendix K: \"we test the model on its understanding of Switzerland's environment by developing a novel benchmark SwitzerlandQA specifically tailored to Switzerland's context. [...] The benchmark represents 26 cantons, with each canton having at least 200 questions, yielding 9167 unique items per language across domains and levels of granularity.\" The artefact list names it as swiss-ai/switzerland_qa, but no public URL was found - see notes."
            },
            {
              "name": "apertus-pretrain-toxicity",
              "kind": "model",
              "url": "https://huggingface.co/swiss-ai/apertus-pretrain-toxicity",
              "evidence": "Model card: \"Language specific toxicity classifiers in English, French, German, Italian, Spanish, Portuguese, Polish, Chinese and Dutch, trained on PleIAs/ToxicCommons and SWSR-SexComments datasets.\" and \"The classifier checkpoints with the best accuracy on the held-out validation set are further employed to annotate the toxicity scores on FineWeb-2 and FineWeb.\" The paper cites this exact URL as footnote 13 in the toxicity-filtering section."
            },
            {
              "name": "RealToxicityPrompts-Llama-Subsampled",
              "kind": "benchmark",
              "url": "https://huggingface.co/datasets/swiss-ai/realtoxicityprompts",
              "evidence": "Paper: \"To integrate it in our benchmark harness, we sub-sample it to 10% of its size and switch the toxicity classifier model to Llama-Guard-3-8B (Fedorov et al., 2024) to allow fully-contained execution. We release this subsample, [48] as well as the LLaMA-Guard-3-8B implementation. [49] The resulting benchmark, RealToxicityPromptsLlama-Subsampled...\" with footnote 48 = https://huggingface.co/datasets/swiss-ai/realtoxicityprompts/tree/main/realtoxicityprompts_small. The dataset page confirms two subsets, realtoxicityprompts_full (99.4k rows) and realtoxicityprompts_small (10k rows)."
            }
          ],
          "follow_up": [
            {
              "what": "Swisscom deployed Apertus on its sovereign Swiss AI Platform for business customers on the day of release",
              "kind": "deployment",
              "url": "https://www.swisscom.ch/en/about/news/2025/09/02-apertus.html",
              "evidence": "Swisscom is proud to be among the first to deploy this pioneering large language model on our sovereign Swiss AI Platform. [...] As of today, Swisscom business customers will be able to access the Apertus model via Swisscom's sovereign Swiss AI platform."
            },
            {
              "what": "The Public AI Inference Utility became the official international deployer of Apertus, serving it worldwide",
              "kind": "deployment",
              "url": "https://publicai.co/stories/apertus",
              "evidence": "Public AI is proud to be the official international deployer for Apertus. To support Apertus, we've allocated over 115000 GPU-hours spread across 20 clusters in 5+ countries - just for the month of September."
            },
            {
              "what": "Apertus v1.1, a family of distilled models trained from the Apertus-8B-2509 teacher described in this report",
              "kind": "model",
              "url": "https://huggingface.co/swiss-ai/Apertus-v1.1-4B-Instruct",
              "evidence": "For more details refer to the original Apertus [technical report](https://arxiv.org/abs/2509.14233) and the new Apertus [distillation technical report](https://arxiv.org/abs/2605.29128)."
            },
            {
              "what": "Apertus v1.5, a multimodal continuation of the models released in this report",
              "kind": "successor-work",
              "url": "https://huggingface.co/swiss-ai/Apertus-v1.5-70B",
              "evidence": "The released models are the result of continued pretraining of Apertus 1.0, adding a multimodal mix of 4T tokens to the 8B model and 2T tokens to the 70B model."
            }
          ],
          "data_description": "15T tokens from 1800+ languages, filtered for toxicity and compliance.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "Switzerland"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Safety alignment",
              "Misuse risk assessment",
              "Constitutional AI"
            ],
            "Toxicity & harassment": [
              "Model-generated toxicity",
              "Implicit hate speech"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Misuse risk assessment",
              "Safety alignment"
            ],
            "Toxicity & harassment": [
              "Implicit hate speech",
              "Model-generated toxicity"
            ]
          },
          "tentative": false
        },
        {
          "title": "\"Flex Tape Can't Fix That\": Bias and Misinformation in Edited Language Models",
          "wid": "flex-tape-can-t-fix-that-bias-and-misinformation-in-edi",
          "type": "publication",
          "year": 2024,
          "venue": "EMNLP 2024 (Main Conference)",
          "link": "https://aclanthology.org/2024.emnlp-main.494/",
          "authors": [
            "Karina Halevy (EPFL / Carnegie Mellon)",
            "Anna Sotnikova (EPFL / U Maryland)",
            "Badr AlKhamissi (EPFL NLP)",
            "Syrielle Montariol (EPFL NLP)",
            "Antoine Bosselut (EPFL NLP)"
          ],
          "epfl_authors": [
            "Karina Halevy (EPFL / Carnegie Mellon)",
            "Anna Sotnikova (EPFL / U Maryland)",
            "Badr AlKhamissi (EPFL NLP)",
            "Syrielle Montariol (EPFL NLP)",
            "Antoine Bosselut (EPFL NLP)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "H"
          ],
          "mdh_topics": [
            "Model editing",
            "LLM bias",
            "Misinformation",
            "Sexism",
            "Xenophobia",
            "Fairness"
          ],
          "stage": "Monitoring",
          "relevance": 4,
          "about": "Study of a hidden cost of editing facts directly into a language model's weights: the edits leak into unrelated knowledge and amplify demographic bias. The authors build a benchmark of knowledge edits across properties such as gender, citizenship and birthplace, compare three editing methods across five models, and have annotators score the open-ended generations. All three methods amplify bias, with especially large confidence drops for Asian, African and Middle Eastern subjects and significant rises in sexism after gender edits and in xenophobia and racism after citizenship edits. Weight-based editing can pass standard specificity tests yet still inject misinformation and worsen bias against marginalised groups.",
          "why": "It shows that editing LLMs injects misinformation and amplifies sexism, xenophobia and racism against marginalised groups.",
          "data": "Text (SeeSaw-CF benchmark, 3516 edits, 734620 cloze prompts, 27010 open-ended prompts)",
          "themes": [
            "AI safety",
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Bias amplification",
            "Factual robustness",
            "Model-generated toxicity",
            "Safety alignment",
            "Sexism",
            "Xenophobia"
          ],
          "key_terms": [
            "Model editing",
            "Bias amplification",
            "Demographic bias",
            "SEESAW-CF",
            "LLM safety"
          ],
          "models": [
            "GPT-3.5",
            "GPT-J",
            "Llama 2",
            "Llama2",
            "Mistral"
          ],
          "method_qualifiers": [
            "MEMIT",
            "Model editing",
            "Manual annotation",
            "Cloze-completion probing"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "SEESAW-CF",
              "kind": "benchmark",
              "url": "https://github.com/ENSCMA2/flextape",
              "evidence": "Paper: 'We release our code and data publicly. 3' with footnote '3 https://github.com/ENSCMA2/flextape'. The repository README states: 'Here is the code used for the paper _\"Flex Tape Can't Fix That\": Pitfalls of Model Editing_, accepted to EMNLP 2024 (preprint https://arxiv.org/abs/2403.00180).' Its data/ directory holds the benchmark itself as seesaw_cf_P101.json, seesaw_cf_P103.json, seesaw_cf_P19_P101.json, seesaw_cf_P21_P101.json, seesaw_cf_P27_P101.json and further seesaw_cf_* files, matching the README's instruction that SEESAW-CF files 'begin with seesaw_cf_'."
            }
          ],
          "data_description": "3516 edits, 734k cloze prompts, 27k open-ended prompts across 5 LLMs.",
          "follow_up_checked": "2026-09-02",
          "targeted_group": [
            "African people",
            "Black people",
            "East Asian people",
            "Jewish people",
            "Middle Eastern people",
            "Transgender women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Bias amplification",
              "Factual robustness",
              "Safety alignment"
            ],
            "Toxicity & harassment": [
              "Model-generated toxicity",
              "Sexism",
              "Xenophobia"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Bias amplification",
              "Factual robustness",
              "Safety alignment"
            ],
            "Toxicity & harassment": [
              "Identity-targeted hate",
              "Model-generated toxicity"
            ]
          },
          "tentative": false
        },
        {
          "title": "Fine-tuning and Sampling Strategies for Multimodal Role Labeling of Entities under Class Imbalance",
          "wid": "fine-tuning-and-sampling-strategies-for-multimodal-role",
          "type": "publication",
          "year": 2022,
          "venue": "Workshop on Combating Online Hostile Posts in Regional Languages during Emergency Situations, ACL 2022",
          "link": "https://aclanthology.org/2022.constraint-1.7/",
          "authors": [
            "Syrielle Montariol",
            "Étienne Simon",
            "Arij Riabi",
            "Djamé Seddah"
          ],
          "about": "Written at INRIA Paris, before Syrielle Montariol joined EPFL. A system for the CONSTRAINT'22 shared task of classifying the role of entities in memes as hero, villain, victim or other from the perspective of the meme's author, on English memes about COVID-19 and US politics. It pairs pre-trained multimodal encoders (CLIP, VisualBERT and OFA) with three classifier designs and tackles class imbalance, with \"other\" making up 78 percent of the training set, through sampling strategies. Dynamic sampling outperformed static sampling, CLIP with an MLP classifier was the best standalone system (47.0 macro-F1), and the best ensemble reached 47.9. Five human annotators averaged 65.5 macro-F1 on 100 meme-entity pairs, 25 per role, against a top challenge score of 58.7.",
          "themes": [
            "Media framing & narrative analysis"
          ],
          "subtopics": [
            "Frame identification",
            "Issue framing",
            "Semantic role labeling"
          ],
          "mdh_focus": [
            "D",
            "H"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 4,
          "stage": "Monitoring",
          "key_terms": [
            "Class imbalance",
            "COVID-19",
            "US politics",
            "Memes",
            "CONSTRAINT shared task"
          ],
          "models": [
            "CLIP",
            "OFA",
            "VisualBERT"
          ],
          "method_qualifiers": [
            "Model ensembling",
            "Dynamic sampling",
            "Manual annotation",
            "Cross-modal attention"
          ],
          "events_cases": [
            "COVID-19 pandemic"
          ],
          "built_at_epfl": [],
          "data_description": "17.5k train, 2k val, 2.4k test meme-entity pairs; COVID-19 and US politics.",
          "region_country": [
            "United States"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Media framing & narrative analysis": [
              "Frame identification",
              "Issue framing",
              "Semantic role labeling"
            ]
          },
          "theme_qualifiers_canonical": {
            "Media framing & narrative analysis": [
              "Frame identification",
              "Issue framing"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "anna-sotnikova",
      "name": "Anna Sotnikova",
      "url": "https://people.epfl.ch/anna.sotnikova",
      "unit": "Centre for Digital Education (CEDE)",
      "faculty": "EPFL",
      "mdh_focus": [],
      "dataTypes": [],
      "techTypes": [],
      "stage": "Prevention + Monitoring",
      "publications": [
        {
          "title": "Analyzing Stereotypes in Generative Text Inference Tasks",
          "wid": "analyzing-stereotypes-in-generative-text-inference-task",
          "type": "publication",
          "year": 2021,
          "venue": "Findings of ACL-IJCNLP 2021",
          "link": "https://aclanthology.org/2021.findings-acl.355/",
          "authors": [
            "Anna Sotnikova",
            "Yang Trista Cao",
            "Hal Daumé III",
            "Rachel Rudinger"
          ],
          "about": "Written at the University of Maryland, before Anna Sotnikova joined EPFL. Study of stereotypes in the hypotheses that generative text inference models produce from neutral, real-life premises naming a person's social category. Hypotheses come from GPT-2 models finetuned on SNLI and MNLI and from COMET, covering 71 target category terms (61 categories) in six stereotype domains, with human judgments on 1281 examples. The models generate more stereotyped hypotheses for socioeconomic status, politics and nationality than for race, gender and religion, and hypotheses about formerly incarcerated, poor, working class and Filipino people are highly dependent on identity. Annotators largely agree on whether hypotheses are valid and plausible, but disagree on identity, sentiment and stereotypes.",
          "themes": [
            "AI safety",
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Annotator positionality",
            "Bias amplification",
            "Contested definitions",
            "Misuse risk assessment",
            "Model-generated toxicity",
            "Representational harm"
          ],
          "mdh_focus": [
            "H"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "Monitoring",
          "key_terms": [
            "Natural language inference",
            "Commonsense inference",
            "Annotator positionality",
            "Representational harm",
            "Stereotypes"
          ],
          "models": [
            "COMET",
            "GPT-2"
          ],
          "method_qualifiers": [
            "Manual annotation",
            "Template-based probing"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Annotated stereotype generative-inference examples",
              "kind": "dataset"
            },
            {
              "name": "Stereotypes Generative Inferences Dataset",
              "kind": "dataset"
            }
          ],
          "data_description": "130k generated premise-hypothesis pairs, 1281 human-annotated.",
          "region_country": [
            "United States"
          ],
          "targeted_group": [
            "Ethnic minorities",
            "Formerly incarcerated people",
            "Gender",
            "Immigrants",
            "Low-income and working-class people",
            "Low-income groups",
            "Migrants",
            "Nationality",
            "Political affiliation",
            "Race",
            "Racial minorities",
            "Religion",
            "Religious minorities",
            "Socioeconomic status",
            "Transgender people",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Bias amplification",
              "Representational harm",
              "Misuse risk assessment"
            ],
            "Toxicity & harassment": [
              "Model-generated toxicity",
              "Annotator positionality",
              "Contested definitions"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Bias amplification",
              "Misuse risk assessment"
            ],
            "Toxicity & harassment": [
              "Contested definitions",
              "Model-generated toxicity"
            ]
          },
          "tentative": false
        },
        {
          "title": "Theory-Grounded Measurement of U.S. Social Stereotypes in English Language Models",
          "wid": "theory-grounded-measurement-of-u-s-social-stereotypes-i",
          "type": "publication",
          "year": 2022,
          "venue": "NAACL 2022",
          "link": "https://aclanthology.org/2022.naacl-main.92/",
          "authors": [
            "Yang Trista Cao",
            "Anna Sotnikova",
            "Hal Daumé III",
            "Rachel Rudinger",
            "Linda Zou"
          ],
          "about": "Written at the University of Maryland, before Anna Sotnikova joined EPFL. Study measuring social stereotypes in English masked language models, grounded in the Agency-Belief-Communion (ABC) stereotype model from social psychology, which covers 16 trait pairs in three dimensions. It introduces the sensitivity test (SeT), which computes the minimal change to a model's last layer needed to make a given trait the most probable, and compares it with other measures against group-trait judgments collected from U.S.-based participants for 25 social groups. SeT with RoBERTa aligns most with human judgments (Kendall's tau 0.199), which the authors call moderate correlation. For intersectional groups, age and political stance are generally dominant domains in the model, while race and nationality are dominated.",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Bias amplification",
            "Misuse risk assessment",
            "Stereotype measurement"
          ],
          "mdh_focus": [
            "H"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 2,
          "stage": "NA",
          "key_terms": [
            "ABC stereotype model",
            "Intersectional stereotypes",
            "Language model bias",
            "ABC model",
            "Agency-Belief-Communion model"
          ],
          "models": [
            "BERT",
            "RoBERTa"
          ],
          "method_qualifiers": [
            "Manual annotation",
            "Zero-shot prompting",
            "Template-based probing",
            "Human judgment alignment"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Sensitivity test (SeT)",
              "kind": "tool"
            },
            {
              "name": "U.S_Stereotypes dataset",
              "kind": "dataset"
            },
            {
              "name": "U.S. group-trait human judgment dataset",
              "kind": "dataset"
            }
          ],
          "data_description": "Human judgments from 133 U.S. participants on 25 social groups.",
          "targeted_group": [
            "Age",
            "Asian people",
            "Black people",
            "Christian people",
            "Gender and sexuality",
            "Hispanic people",
            "LGBTQ+",
            "Marginalized groups",
            "Men",
            "Muslim people",
            "Nationality",
            "People with disabilities",
            "Political affiliation",
            "Race and ethnicity",
            "Racial and ethnic minorities",
            "Religion",
            "Religious minorities",
            "Socioeconomic status",
            "White people",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Bias amplification",
              "Misuse risk assessment",
              "Stereotype measurement"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Bias amplification",
              "Misuse risk assessment"
            ]
          },
          "tentative": false
        },
        {
          "title": "Which Examples Should be Multiply Annotated? Active Learning When Annotators May Disagree",
          "wid": "which-examples-should-be-multiply-annotated-active-lear",
          "type": "publication",
          "year": 2023,
          "venue": "Findings of ACL 2023",
          "link": "https://aclanthology.org/2023.findings-acl.658/",
          "authors": [
            "Connor Baumler",
            "Anna Sotnikova",
            "Hal Daumé III"
          ],
          "about": "Written at the University of Maryland, before Anna Sotnikova joined EPFL. Paper developing Disagreement Aware Active Learning (DAAL), an active learning approach for training classifiers that predict full label distributions on tasks where annotators may disagree, such as hate speech and toxicity detection. It trains an entropy predictor on very few multiply-labeled examples and queries the examples where model entropy and estimated annotator entropy differ most. In simulations on the Measuring Hate Speech and Wikipedia Talk datasets, uncertainty-based active learning underperforms passive learning on high-disagreement tasks, while DAAL needs at least 24 percent fewer annotations on average than the strongest competitor, with one exception on the Toxicity task.",
          "themes": [
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Comment toxicity scoring",
            "Contested definitions",
            "Soft-label prediction"
          ],
          "mdh_focus": [
            "H"
          ],
          "mdh_relevance": "direct",
          "relevance": 4,
          "stage": "Monitoring",
          "key_terms": [
            "Annotator disagreement",
            "Active learning",
            "Hate speech detection",
            "Toxicity detection",
            "Annotation cost"
          ],
          "models": [
            "RoBERTa-base"
          ],
          "method_qualifiers": [
            "Active learning",
            "Soft-label training",
            "Uncertainty sampling"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "DAAL",
              "kind": "tool"
            },
            {
              "name": "DAAL (Disagreement Aware Active Learning)",
              "kind": "framework"
            }
          ],
          "data_description": "MHS and Wikipedia Toxicity datasets, ~20k-50k examples, 3-10 annotations per example.",
          "platform": [
            "Reddit",
            "Twitter/X",
            "Wikipedia",
            "YouTube"
          ],
          "targeted_group": [
            "LGBTQ+"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Toxicity & harassment": [
              "Contested definitions",
              "Comment toxicity scoring",
              "Soft-label prediction"
            ]
          },
          "theme_qualifiers_canonical": {
            "Toxicity & harassment": [
              "Comment toxicity scoring",
              "Contested definitions"
            ]
          },
          "tentative": false
        },
        {
          "title": "Apertus: Democratizing Open and Compliant LLMs for Global Language Environments",
          "wid": "apertus-democratizing-open-and-compliant-llms-for-globa",
          "type": "publication",
          "year": 2025,
          "venue": "arXiv 2509.14233",
          "link": "https://arxiv.org/abs/2509.14233",
          "authors": [
            "Alejandro Hernández-Cano",
            "Alexander Hägele",
            "Allen Hao Huang",
            "Angelika Romanou",
            "Antoni-Joan Solergibert",
            "Barna Pasztor",
            "Bettina Messmer",
            "Dhia Garbaya",
            "Eduard Frank Ďurech",
            "Ido Hakimi",
            "Juan García Giraldo",
            "Mete Ismayilzada",
            "Negar Foroutan",
            "Skander Moalla",
            "Tiancheng Chen",
            "Vinko Sabolčec",
            "Yixuan Xu",
            "Michael Aerni",
            "Badr AlKhamissi",
            "Inés Altemir Mariñas",
            "Mohammad Hossein Amani",
            "Matin Ansaripour",
            "Ilia Badanin",
            "Harold Benoit",
            "Emanuela Boros",
            "Nicholas Browning",
            "Fabian Bösch",
            "Maximilian Böther",
            "Niklas Canova",
            "Camille Challier",
            "Clement Charmillot",
            "Jonathan Coles",
            "Jan Deriu",
            "Arnout Devos",
            "Lukas Drescher",
            "Daniil Dzenhaliou",
            "Maud Ehrmann",
            "Dongyang Fan",
            "Simin Fan",
            "Silin Gao",
            "Miguel Gila",
            "María Grandury",
            "Diba Hashemi",
            "Alexander Hoyle",
            "Jiaming Jiang",
            "Mark Klein",
            "Andrei Kucharavy",
            "Anastasiia Kucherenko",
            "Frederike Lübeck",
            "Roman Machacek",
            "Theofilos Manitaras",
            "Andreas Marfurt",
            "Kyle Matoba",
            "Simon Matrenok",
            "Henrique Mendonça",
            "Fawzi Roberto Mohamed",
            "Syrielle Montariol",
            "Luca Mouchel",
            "Sven Najem-Meyer",
            "Jingwei Ni",
            "Gennaro Oliva",
            "Matteo Pagliardini",
            "Elia Palme",
            "Andrei Panferov",
            "Léo Paoletti",
            "Marco Passerini",
            "Ivan Pavlov",
            "Auguste Poiroux",
            "Kaustubh Ponkshe",
            "Nathan Ranchin",
            "Javi Rando",
            "Mathieu Sauser",
            "Jakhongir Saydaliev",
            "Muhammad Ali Sayfiddinov",
            "Marian Schneider",
            "Stefano Schuppli",
            "Marco Scialanga",
            "Andrei Semenov",
            "Kumar Shridhar",
            "Raghav Singhal",
            "Anna Sotnikova",
            "Alexander Sternfeld",
            "Ayush Kumar Tarun",
            "Paul Teiletche",
            "Jannis Vamvas",
            "Xiaozhe Yao",
            "Hao Zhao",
            "Alexander Ilic",
            "Ana Klimovic",
            "Andreas Krause",
            "Caglar Gulcehre",
            "David Rosenthal",
            "Elliott Ash",
            "Florian Tramèr",
            "Joost VandeVondele",
            "Livio Veraldi",
            "Martin Rajman",
            "Thomas Schulthess",
            "Torsten Hoefler",
            "Antoine Bosselut",
            "Martin Jaggi",
            "Imanol Schlag"
          ],
          "epfl_authors": [
            "Alejandro Hernández-Cano",
            "Alexander Hägele",
            "Allen Hao Huang",
            "Angelika Romanou",
            "Antoni-Joan Solergibert",
            "Bettina Messmer",
            "Dhia Garbaya",
            "Eduard Frank Ďurech",
            "Juan García Giraldo",
            "Mete Ismayilzada",
            "Negar Foroutan",
            "Skander Moalla",
            "Vinko Sabolčec",
            "Yixuan Xu",
            "Badr AlKhamissi",
            "Inés Altemir Mariñas",
            "Mohammad Hossein Amani",
            "Matin Ansaripour",
            "Ilia Badanin",
            "Harold Benoit",
            "Emanuela Boros",
            "Camille Challier",
            "Clement Charmillot",
            "Daniil Dzenhaliou",
            "Maud Ehrmann",
            "Dongyang Fan",
            "Simin Fan",
            "Silin Gao",
            "María Grandury",
            "Diba Hashemi",
            "Jiaming Jiang",
            "Kyle Matoba",
            "Simon Matrenok",
            "Syrielle Montariol",
            "Luca Mouchel",
            "Sven Najem-Meyer",
            "Matteo Pagliardini",
            "Léo Paoletti",
            "Ivan Pavlov",
            "Auguste Poiroux",
            "Kaustubh Ponkshe",
            "Nathan Ranchin",
            "Mathieu Sauser",
            "Jakhongir Saydaliev",
            "Marco Scialanga",
            "Andrei Semenov",
            "Raghav Singhal",
            "Anna Sotnikova",
            "Ayush Kumar Tarun",
            "Paul Teiletche",
            "Hao Zhao",
            "Caglar Gulcehre",
            "Martin Rajman",
            "Antoine Bosselut",
            "Martin Jaggi"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_topics": [
            "Open-source LLM",
            "Data compliance",
            "Multilingual",
            "Toxic-content filtering",
            "Swiss AI Initiative",
            "Digital sovereignty"
          ],
          "stage": "Prevention + Monitoring",
          "relevance": 3,
          "about": "Technical report for Apertus, a fully open suite of large language models trained on web-scale multilingual data spanning more than 1800 languages, with around 40 percent of pretraining data non-English. It trains only on openly available data, respects robots.txt exclusions, and filters out non-permissive, toxic and personally identifiable content, with everything released under a permissive license so the pipeline can be audited. Released at 8B and 70B scales and trained on 15 trillion tokens, it approaches state-of-the-art results among fully open models on multilingual benchmarks while training only on compliant data.",
          "why": "Its toxic-content filtering and auditable, opt-out-respecting pretraining make it a trustworthy base model for downstream MDH work.",
          "data": "Text, 15 trillion tokens across 1800+ languages",
          "lab": [
            "NLP",
            "MLO",
            "CLAIRE"
          ],
          "themes": [
            "AI safety",
            "Platform governance & regulation",
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Constitutional AI",
            "Implicit hate speech",
            "Misuse risk assessment",
            "Model-generated toxicity",
            "Safety alignment"
          ],
          "key_terms": [
            "Data compliance",
            "Swiss AI Charter",
            "Constitutional AI",
            "Constitutional AI alignment",
            "EU AI Act compliance"
          ],
          "models": [
            "Apertus-70B-Instruct"
          ],
          "method_qualifiers": [
            "Goldfish loss",
            "Constitutional AI",
            "Preference optimisation",
            "QRPO alignment"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Apertus-70B-Instruct",
              "kind": "model"
            },
            {
              "name": "Apertus-sft-mixture",
              "kind": "dataset"
            },
            {
              "name": "Swiss AI Charter",
              "kind": "framework",
              "url": "https://www.apertus-ai.org/pages/charter/",
              "evidence": "The Apertus Charter defines alignment principles for AI systems developed under the Swiss AI Initiative, rooted in Switzerland's constitutional values and democratic traditions. (Previously referred to as the Swiss AI Charter) Version 1.0 - August 2025"
            },
            {
              "name": "SwitzerlandQA",
              "kind": "benchmark",
              "evidence": "Paper Appendix K: \"we test the model on its understanding of Switzerland's environment by developing a novel benchmark SwitzerlandQA specifically tailored to Switzerland's context. [...] The benchmark represents 26 cantons, with each canton having at least 200 questions, yielding 9167 unique items per language across domains and levels of granularity.\" The artefact list names it as swiss-ai/switzerland_qa, but no public URL was found - see notes."
            },
            {
              "name": "apertus-pretrain-toxicity",
              "kind": "model",
              "url": "https://huggingface.co/swiss-ai/apertus-pretrain-toxicity",
              "evidence": "Model card: \"Language specific toxicity classifiers in English, French, German, Italian, Spanish, Portuguese, Polish, Chinese and Dutch, trained on PleIAs/ToxicCommons and SWSR-SexComments datasets.\" and \"The classifier checkpoints with the best accuracy on the held-out validation set are further employed to annotate the toxicity scores on FineWeb-2 and FineWeb.\" The paper cites this exact URL as footnote 13 in the toxicity-filtering section."
            },
            {
              "name": "RealToxicityPrompts-Llama-Subsampled",
              "kind": "benchmark",
              "url": "https://huggingface.co/datasets/swiss-ai/realtoxicityprompts",
              "evidence": "Paper: \"To integrate it in our benchmark harness, we sub-sample it to 10% of its size and switch the toxicity classifier model to Llama-Guard-3-8B (Fedorov et al., 2024) to allow fully-contained execution. We release this subsample, [48] as well as the LLaMA-Guard-3-8B implementation. [49] The resulting benchmark, RealToxicityPromptsLlama-Subsampled...\" with footnote 48 = https://huggingface.co/datasets/swiss-ai/realtoxicityprompts/tree/main/realtoxicityprompts_small. The dataset page confirms two subsets, realtoxicityprompts_full (99.4k rows) and realtoxicityprompts_small (10k rows)."
            }
          ],
          "follow_up": [
            {
              "what": "Swisscom deployed Apertus on its sovereign Swiss AI Platform for business customers on the day of release",
              "kind": "deployment",
              "url": "https://www.swisscom.ch/en/about/news/2025/09/02-apertus.html",
              "evidence": "Swisscom is proud to be among the first to deploy this pioneering large language model on our sovereign Swiss AI Platform. [...] As of today, Swisscom business customers will be able to access the Apertus model via Swisscom's sovereign Swiss AI platform."
            },
            {
              "what": "The Public AI Inference Utility became the official international deployer of Apertus, serving it worldwide",
              "kind": "deployment",
              "url": "https://publicai.co/stories/apertus",
              "evidence": "Public AI is proud to be the official international deployer for Apertus. To support Apertus, we've allocated over 115000 GPU-hours spread across 20 clusters in 5+ countries - just for the month of September."
            },
            {
              "what": "Apertus v1.1, a family of distilled models trained from the Apertus-8B-2509 teacher described in this report",
              "kind": "model",
              "url": "https://huggingface.co/swiss-ai/Apertus-v1.1-4B-Instruct",
              "evidence": "For more details refer to the original Apertus [technical report](https://arxiv.org/abs/2509.14233) and the new Apertus [distillation technical report](https://arxiv.org/abs/2605.29128)."
            },
            {
              "what": "Apertus v1.5, a multimodal continuation of the models released in this report",
              "kind": "successor-work",
              "url": "https://huggingface.co/swiss-ai/Apertus-v1.5-70B",
              "evidence": "The released models are the result of continued pretraining of Apertus 1.0, adding a multimodal mix of 4T tokens to the 8B model and 2T tokens to the 70B model."
            }
          ],
          "data_description": "15T tokens from 1800+ languages, filtered for toxicity and compliance.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "Switzerland"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Safety alignment",
              "Misuse risk assessment",
              "Constitutional AI"
            ],
            "Toxicity & harassment": [
              "Model-generated toxicity",
              "Implicit hate speech"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Misuse risk assessment",
              "Safety alignment"
            ],
            "Toxicity & harassment": [
              "Implicit hate speech",
              "Model-generated toxicity"
            ]
          },
          "tentative": false
        },
        {
          "title": "\"Flex Tape Can't Fix That\": Bias and Misinformation in Edited Language Models",
          "wid": "flex-tape-can-t-fix-that-bias-and-misinformation-in-edi",
          "type": "publication",
          "year": 2024,
          "venue": "EMNLP 2024 (Main Conference)",
          "link": "https://aclanthology.org/2024.emnlp-main.494/",
          "authors": [
            "Karina Halevy (EPFL / Carnegie Mellon)",
            "Anna Sotnikova (EPFL / U Maryland)",
            "Badr AlKhamissi (EPFL NLP)",
            "Syrielle Montariol (EPFL NLP)",
            "Antoine Bosselut (EPFL NLP)"
          ],
          "epfl_authors": [
            "Karina Halevy (EPFL / Carnegie Mellon)",
            "Anna Sotnikova (EPFL / U Maryland)",
            "Badr AlKhamissi (EPFL NLP)",
            "Syrielle Montariol (EPFL NLP)",
            "Antoine Bosselut (EPFL NLP)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "H"
          ],
          "mdh_topics": [
            "Model editing",
            "LLM bias",
            "Misinformation",
            "Sexism",
            "Xenophobia",
            "Fairness"
          ],
          "stage": "Monitoring",
          "relevance": 4,
          "about": "Study of a hidden cost of editing facts directly into a language model's weights: the edits leak into unrelated knowledge and amplify demographic bias. The authors build a benchmark of knowledge edits across properties such as gender, citizenship and birthplace, compare three editing methods across five models, and have annotators score the open-ended generations. All three methods amplify bias, with especially large confidence drops for Asian, African and Middle Eastern subjects and significant rises in sexism after gender edits and in xenophobia and racism after citizenship edits. Weight-based editing can pass standard specificity tests yet still inject misinformation and worsen bias against marginalised groups.",
          "why": "It shows that editing LLMs injects misinformation and amplifies sexism, xenophobia and racism against marginalised groups.",
          "data": "Text (SeeSaw-CF benchmark, 3516 edits, 734620 cloze prompts, 27010 open-ended prompts)",
          "themes": [
            "AI safety",
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Bias amplification",
            "Factual robustness",
            "Model-generated toxicity",
            "Safety alignment",
            "Sexism",
            "Xenophobia"
          ],
          "key_terms": [
            "Model editing",
            "Bias amplification",
            "Demographic bias",
            "SEESAW-CF",
            "LLM safety"
          ],
          "models": [
            "GPT-3.5",
            "GPT-J",
            "Llama 2",
            "Llama2",
            "Mistral"
          ],
          "method_qualifiers": [
            "MEMIT",
            "Model editing",
            "Manual annotation",
            "Cloze-completion probing"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "SEESAW-CF",
              "kind": "benchmark",
              "url": "https://github.com/ENSCMA2/flextape",
              "evidence": "Paper: 'We release our code and data publicly. 3' with footnote '3 https://github.com/ENSCMA2/flextape'. The repository README states: 'Here is the code used for the paper _\"Flex Tape Can't Fix That\": Pitfalls of Model Editing_, accepted to EMNLP 2024 (preprint https://arxiv.org/abs/2403.00180).' Its data/ directory holds the benchmark itself as seesaw_cf_P101.json, seesaw_cf_P103.json, seesaw_cf_P19_P101.json, seesaw_cf_P21_P101.json, seesaw_cf_P27_P101.json and further seesaw_cf_* files, matching the README's instruction that SEESAW-CF files 'begin with seesaw_cf_'."
            }
          ],
          "data_description": "3516 edits, 734k cloze prompts, 27k open-ended prompts across 5 LLMs.",
          "follow_up_checked": "2026-09-02",
          "targeted_group": [
            "African people",
            "Black people",
            "East Asian people",
            "Jewish people",
            "Middle Eastern people",
            "Transgender women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Bias amplification",
              "Factual robustness",
              "Safety alignment"
            ],
            "Toxicity & harassment": [
              "Model-generated toxicity",
              "Sexism",
              "Xenophobia"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Bias amplification",
              "Factual robustness",
              "Safety alignment"
            ],
            "Toxicity & harassment": [
              "Identity-targeted hate",
              "Model-generated toxicity"
            ]
          },
          "tentative": false
        },
        {
          "title": "Multilingual large language models leak human stereotypes across language boundaries",
          "wid": "multilingual-large-language-models-leak-human-stereotyp",
          "type": "publication",
          "year": 2024,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2312.07141",
          "authors": [
            "Yang Trista Cao",
            "Anna Sotnikova",
            "Jieyu Zhao",
            "Linda X. Zou",
            "Rachel Rudinger",
            "Hal Daumé III"
          ],
          "about": "Anna Sotnikova's affiliation on this paper is the University of Maryland, not EPFL. A study of stereotype leakage, where training a model multilingually may lead stereotypes expressed in one language to appear in its behavior in another. It proposes a measurement framework, applied to GPT-3.5, mT5 and mBERT in English, Russian, Chinese and Hindi, using stereotypes about 30 social groups collected from native speakers. All three models show leakage, and languages transmitting stereotypes likely also receive some; GPT-3.5 shows the most (seven significant leakages) and Hindi receives the most, possibly because it is the only low-resource language tested. In GPT-3.5, leaked associations are positive, negative and non-polar, and stereotypes of groups unknown to other linguistic communities carry over from their language of origin.",
          "themes": [
            "AI safety"
          ],
          "subtopics": [
            "Bias amplification",
            "Cross-lingual stereotype leakage"
          ],
          "mdh_focus": [
            "H"
          ],
          "mdh_relevance": "adjacent",
          "relevance": 3,
          "stage": "Monitoring",
          "key_terms": [
            "Stereotype leakage",
            "Cross-lingual bias transfer",
            "Anglocentrism",
            "Multilingual LLMs",
            "Cross-lingual bias"
          ],
          "models": [
            "GPT-3.5",
            "mBERT",
            "mT5"
          ],
          "method_qualifiers": [
            "Mixed-effects regression",
            "Cross-lingual comparison",
            "Log-probability probing"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Multilingual stereotype leakage dataset and code",
              "kind": "dataset"
            },
            {
              "name": "Stereotype leakage measurement framework",
              "kind": "framework"
            }
          ],
          "data_description": "286 Prolific participants, 4 languages, 30 social groups, 3 MLLMs.",
          "region_country": [
            "China",
            "India",
            "Russia",
            "United States"
          ],
          "targeted_group": [
            "Asian people",
            "Black people",
            "Caste-based groups",
            "Feminists",
            "Immigrants",
            "LGBTQ+ people",
            "Marginalized groups",
            "Religious and ethnic minorities",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Bias amplification",
              "Cross-lingual stereotype leakage"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Bias amplification"
            ]
          },
          "tentative": false
        }
      ]
    },
    {
      "id": "badr-alkhamissi",
      "name": "Badr AlKhamissi",
      "url": "https://people.epfl.ch/badr.alkhamissi",
      "unit": "NLP Lab and Schrimpf Group",
      "faculty": "IC / SV",
      "mdh_focus": [],
      "dataTypes": [],
      "techTypes": [],
      "stage": "Prevention + Monitoring",
      "publications": [
        {
          "title": "Meta AI at Arabic Hate Speech 2022: MultiTask Learning with Self-Correction for Hate Speech Classification",
          "wid": "meta-ai-at-arabic-hate-speech-2022-multitask-learning-w",
          "type": "publication",
          "year": 2022,
          "venue": "arXiv preprint",
          "link": "https://arxiv.org/abs/2205.07960",
          "authors": [
            "Badr AlKhamissi",
            "Mona Diab"
          ],
          "about": "Written at Meta, before Badr AlKhamissi began his PhD at EPFL in 2023. System for the Arabic Fine-Grained Hate Speech Detection shared task, predicting whether a tweet is offensive, whether it is hate speech, and if so which of six fine-grained hate speech categories applies. Working with a dataset of about 13k Arabic tweets, it trains an ensemble of MARBERTv2 models with multitask learning and adds a self-consistency correction that uses the offensive and hate speech predictions to correct the fine-grained label. It reports an F1 macro score of 82.7 percent on the hate speech detection test set, a 3.4 percent relative improvement over previous work, and the correction lowers the fine-grained head's contradiction rate from 2.6 percent to 0.79 percent.",
          "themes": [
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Contested definitions",
            "Identity-targeted hate",
            "Implicit hate speech"
          ],
          "mdh_focus": [
            "H"
          ],
          "mdh_relevance": "direct",
          "relevance": 5,
          "stage": "Monitoring",
          "key_terms": [
            "Arabic hate speech",
            "Self-consistency correction",
            "Multitask learning",
            "Fine-grained classification",
            "MARBERT"
          ],
          "models": [
            "AraHS",
            "MARBERT",
            "MARBERTv2",
            "QARiB"
          ],
          "method_qualifiers": [
            "Multitask learning",
            "Self-consistency correction",
            "Model ensembling",
            "Domain-specific fine-tuning"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "AraHS",
              "kind": "model"
            }
          ],
          "data_description": "~13k Arabic tweets, MSA and Dialectal, 70/10/20 split.",
          "platform": [
            "Twitter"
          ],
          "region_country": [
            "Arab region"
          ],
          "targeted_group": [
            "Disability/Disease",
            "Ethnic minorities",
            "Gender",
            "Ideology",
            "People with disabilities",
            "Race/Ethnicity/Nationality",
            "Religion/Belief",
            "Religious groups",
            "Social Class",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Toxicity & harassment": [
              "Implicit hate speech",
              "Identity-targeted hate",
              "Contested definitions"
            ]
          },
          "theme_qualifiers_canonical": {
            "Toxicity & harassment": [
              "Contested definitions",
              "Identity-targeted hate",
              "Implicit hate speech"
            ]
          },
          "tentative": false
        },
        {
          "title": "ToKen: Task Decomposition and Knowledge Infusion for Few-Shot Hate Speech Detection",
          "wid": "token-task-decomposition-and-knowledge-infusion-for-few",
          "type": "publication",
          "year": 2022,
          "venue": "EMNLP 2022",
          "link": "https://arxiv.org/abs/2205.12495",
          "authors": [
            "Badr AlKhamissi",
            "Faisal Ladhak",
            "Srini Iyer",
            "Ves Stoyanov",
            "Zornitsa Kozareva",
            "Xian Li",
            "Pascale Fung",
            "Lambert Mathias",
            "Asli Celikyilmaz",
            "Mona Diab"
          ],
          "about": "Written at Meta, before Badr AlKhamissi began his PhD at EPFL in 2023. Method, ToKen, for few-shot hate speech detection that decomposes the task: the model predicts whether a post is offensive, whether it targets a group, an individual or neither, and which groups are targeted, before predicting the hate speech label. Trained on few-shot samples from the Social Bias Inference Corpus (SBIC), with BART pre-finetuned on ATOMIC and StereoSet for commonsense reasoning and knowledge of stereotypes, its ATOMIC-infused version beats a baseline predicting the label directly by 17.83 binary F1 points at 16 shots. It consistently outperforms the baseline on three out-of-distribution datasets (significantly on Ethos and HS18) and shows smaller variance across seeds, data partitions and hyperparameters.",
          "themes": [
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Few-shot detection",
            "Implicit hate speech",
            "Implied stereotypes"
          ],
          "mdh_focus": [
            "H"
          ],
          "mdh_relevance": "direct",
          "relevance": 5,
          "stage": "Monitoring",
          "key_terms": [
            "Task decomposition",
            "Knowledge infusion",
            "Few-shot learning",
            "Hate speech detection",
            "ATOMIC 2020"
          ],
          "models": [
            "BART",
            "BART-LARGE",
            "BARTBASE"
          ],
          "method_qualifiers": [
            "Knowledge infusion",
            "Task decomposition",
            "Few-shot learning",
            "Few-shot fine-tuning"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "TOKEN",
              "kind": "framework"
            }
          ],
          "data_description": "SBIC, HateXplain, HS18, Ethos datasets; 16-1024 few-shot samples.",
          "platform": [
            "Gab",
            "Reddit",
            "Stormfront",
            "Twitter/X",
            "YouTube"
          ],
          "targeted_group": [
            "Disabled people",
            "Ethnic minorities",
            "Muslims",
            "Women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "Toxicity & harassment": [
              "Few-shot detection",
              "Implied stereotypes",
              "Implicit hate speech"
            ]
          },
          "theme_qualifiers_canonical": {
            "Toxicity & harassment": [
              "Implicit hate speech"
            ]
          },
          "tentative": false
        },
        {
          "title": "Apertus: Democratizing Open and Compliant LLMs for Global Language Environments",
          "wid": "apertus-democratizing-open-and-compliant-llms-for-globa",
          "type": "publication",
          "year": 2025,
          "venue": "arXiv 2509.14233",
          "link": "https://arxiv.org/abs/2509.14233",
          "authors": [
            "Alejandro Hernández-Cano",
            "Alexander Hägele",
            "Allen Hao Huang",
            "Angelika Romanou",
            "Antoni-Joan Solergibert",
            "Barna Pasztor",
            "Bettina Messmer",
            "Dhia Garbaya",
            "Eduard Frank Ďurech",
            "Ido Hakimi",
            "Juan García Giraldo",
            "Mete Ismayilzada",
            "Negar Foroutan",
            "Skander Moalla",
            "Tiancheng Chen",
            "Vinko Sabolčec",
            "Yixuan Xu",
            "Michael Aerni",
            "Badr AlKhamissi",
            "Inés Altemir Mariñas",
            "Mohammad Hossein Amani",
            "Matin Ansaripour",
            "Ilia Badanin",
            "Harold Benoit",
            "Emanuela Boros",
            "Nicholas Browning",
            "Fabian Bösch",
            "Maximilian Böther",
            "Niklas Canova",
            "Camille Challier",
            "Clement Charmillot",
            "Jonathan Coles",
            "Jan Deriu",
            "Arnout Devos",
            "Lukas Drescher",
            "Daniil Dzenhaliou",
            "Maud Ehrmann",
            "Dongyang Fan",
            "Simin Fan",
            "Silin Gao",
            "Miguel Gila",
            "María Grandury",
            "Diba Hashemi",
            "Alexander Hoyle",
            "Jiaming Jiang",
            "Mark Klein",
            "Andrei Kucharavy",
            "Anastasiia Kucherenko",
            "Frederike Lübeck",
            "Roman Machacek",
            "Theofilos Manitaras",
            "Andreas Marfurt",
            "Kyle Matoba",
            "Simon Matrenok",
            "Henrique Mendonça",
            "Fawzi Roberto Mohamed",
            "Syrielle Montariol",
            "Luca Mouchel",
            "Sven Najem-Meyer",
            "Jingwei Ni",
            "Gennaro Oliva",
            "Matteo Pagliardini",
            "Elia Palme",
            "Andrei Panferov",
            "Léo Paoletti",
            "Marco Passerini",
            "Ivan Pavlov",
            "Auguste Poiroux",
            "Kaustubh Ponkshe",
            "Nathan Ranchin",
            "Javi Rando",
            "Mathieu Sauser",
            "Jakhongir Saydaliev",
            "Muhammad Ali Sayfiddinov",
            "Marian Schneider",
            "Stefano Schuppli",
            "Marco Scialanga",
            "Andrei Semenov",
            "Kumar Shridhar",
            "Raghav Singhal",
            "Anna Sotnikova",
            "Alexander Sternfeld",
            "Ayush Kumar Tarun",
            "Paul Teiletche",
            "Jannis Vamvas",
            "Xiaozhe Yao",
            "Hao Zhao",
            "Alexander Ilic",
            "Ana Klimovic",
            "Andreas Krause",
            "Caglar Gulcehre",
            "David Rosenthal",
            "Elliott Ash",
            "Florian Tramèr",
            "Joost VandeVondele",
            "Livio Veraldi",
            "Martin Rajman",
            "Thomas Schulthess",
            "Torsten Hoefler",
            "Antoine Bosselut",
            "Martin Jaggi",
            "Imanol Schlag"
          ],
          "epfl_authors": [
            "Alejandro Hernández-Cano",
            "Alexander Hägele",
            "Allen Hao Huang",
            "Angelika Romanou",
            "Antoni-Joan Solergibert",
            "Bettina Messmer",
            "Dhia Garbaya",
            "Eduard Frank Ďurech",
            "Juan García Giraldo",
            "Mete Ismayilzada",
            "Negar Foroutan",
            "Skander Moalla",
            "Vinko Sabolčec",
            "Yixuan Xu",
            "Badr AlKhamissi",
            "Inés Altemir Mariñas",
            "Mohammad Hossein Amani",
            "Matin Ansaripour",
            "Ilia Badanin",
            "Harold Benoit",
            "Emanuela Boros",
            "Camille Challier",
            "Clement Charmillot",
            "Daniil Dzenhaliou",
            "Maud Ehrmann",
            "Dongyang Fan",
            "Simin Fan",
            "Silin Gao",
            "María Grandury",
            "Diba Hashemi",
            "Jiaming Jiang",
            "Kyle Matoba",
            "Simon Matrenok",
            "Syrielle Montariol",
            "Luca Mouchel",
            "Sven Najem-Meyer",
            "Matteo Pagliardini",
            "Léo Paoletti",
            "Ivan Pavlov",
            "Auguste Poiroux",
            "Kaustubh Ponkshe",
            "Nathan Ranchin",
            "Mathieu Sauser",
            "Jakhongir Saydaliev",
            "Marco Scialanga",
            "Andrei Semenov",
            "Raghav Singhal",
            "Anna Sotnikova",
            "Ayush Kumar Tarun",
            "Paul Teiletche",
            "Hao Zhao",
            "Caglar Gulcehre",
            "Martin Rajman",
            "Antoine Bosselut",
            "Martin Jaggi"
          ],
          "mdh_relevance": "adjacent",
          "mdh_focus": [
            "M",
            "D",
            "H"
          ],
          "mdh_topics": [
            "Open-source LLM",
            "Data compliance",
            "Multilingual",
            "Toxic-content filtering",
            "Swiss AI Initiative",
            "Digital sovereignty"
          ],
          "stage": "Prevention + Monitoring",
          "relevance": 3,
          "about": "Technical report for Apertus, a fully open suite of large language models trained on web-scale multilingual data spanning more than 1800 languages, with around 40 percent of pretraining data non-English. It trains only on openly available data, respects robots.txt exclusions, and filters out non-permissive, toxic and personally identifiable content, with everything released under a permissive license so the pipeline can be audited. Released at 8B and 70B scales and trained on 15 trillion tokens, it approaches state-of-the-art results among fully open models on multilingual benchmarks while training only on compliant data.",
          "why": "Its toxic-content filtering and auditable, opt-out-respecting pretraining make it a trustworthy base model for downstream MDH work.",
          "data": "Text, 15 trillion tokens across 1800+ languages",
          "lab": [
            "NLP",
            "MLO",
            "CLAIRE"
          ],
          "themes": [
            "AI safety",
            "Platform governance & regulation",
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Constitutional AI",
            "Implicit hate speech",
            "Misuse risk assessment",
            "Model-generated toxicity",
            "Safety alignment"
          ],
          "key_terms": [
            "Data compliance",
            "Swiss AI Charter",
            "Constitutional AI",
            "Constitutional AI alignment",
            "EU AI Act compliance"
          ],
          "models": [
            "Apertus-70B-Instruct"
          ],
          "method_qualifiers": [
            "Goldfish loss",
            "Constitutional AI",
            "Preference optimisation",
            "QRPO alignment"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "Apertus-70B-Instruct",
              "kind": "model"
            },
            {
              "name": "Apertus-sft-mixture",
              "kind": "dataset"
            },
            {
              "name": "Swiss AI Charter",
              "kind": "framework",
              "url": "https://www.apertus-ai.org/pages/charter/",
              "evidence": "The Apertus Charter defines alignment principles for AI systems developed under the Swiss AI Initiative, rooted in Switzerland's constitutional values and democratic traditions. (Previously referred to as the Swiss AI Charter) Version 1.0 - August 2025"
            },
            {
              "name": "SwitzerlandQA",
              "kind": "benchmark",
              "evidence": "Paper Appendix K: \"we test the model on its understanding of Switzerland's environment by developing a novel benchmark SwitzerlandQA specifically tailored to Switzerland's context. [...] The benchmark represents 26 cantons, with each canton having at least 200 questions, yielding 9167 unique items per language across domains and levels of granularity.\" The artefact list names it as swiss-ai/switzerland_qa, but no public URL was found - see notes."
            },
            {
              "name": "apertus-pretrain-toxicity",
              "kind": "model",
              "url": "https://huggingface.co/swiss-ai/apertus-pretrain-toxicity",
              "evidence": "Model card: \"Language specific toxicity classifiers in English, French, German, Italian, Spanish, Portuguese, Polish, Chinese and Dutch, trained on PleIAs/ToxicCommons and SWSR-SexComments datasets.\" and \"The classifier checkpoints with the best accuracy on the held-out validation set are further employed to annotate the toxicity scores on FineWeb-2 and FineWeb.\" The paper cites this exact URL as footnote 13 in the toxicity-filtering section."
            },
            {
              "name": "RealToxicityPrompts-Llama-Subsampled",
              "kind": "benchmark",
              "url": "https://huggingface.co/datasets/swiss-ai/realtoxicityprompts",
              "evidence": "Paper: \"To integrate it in our benchmark harness, we sub-sample it to 10% of its size and switch the toxicity classifier model to Llama-Guard-3-8B (Fedorov et al., 2024) to allow fully-contained execution. We release this subsample, [48] as well as the LLaMA-Guard-3-8B implementation. [49] The resulting benchmark, RealToxicityPromptsLlama-Subsampled...\" with footnote 48 = https://huggingface.co/datasets/swiss-ai/realtoxicityprompts/tree/main/realtoxicityprompts_small. The dataset page confirms two subsets, realtoxicityprompts_full (99.4k rows) and realtoxicityprompts_small (10k rows)."
            }
          ],
          "follow_up": [
            {
              "what": "Swisscom deployed Apertus on its sovereign Swiss AI Platform for business customers on the day of release",
              "kind": "deployment",
              "url": "https://www.swisscom.ch/en/about/news/2025/09/02-apertus.html",
              "evidence": "Swisscom is proud to be among the first to deploy this pioneering large language model on our sovereign Swiss AI Platform. [...] As of today, Swisscom business customers will be able to access the Apertus model via Swisscom's sovereign Swiss AI platform."
            },
            {
              "what": "The Public AI Inference Utility became the official international deployer of Apertus, serving it worldwide",
              "kind": "deployment",
              "url": "https://publicai.co/stories/apertus",
              "evidence": "Public AI is proud to be the official international deployer for Apertus. To support Apertus, we've allocated over 115000 GPU-hours spread across 20 clusters in 5+ countries - just for the month of September."
            },
            {
              "what": "Apertus v1.1, a family of distilled models trained from the Apertus-8B-2509 teacher described in this report",
              "kind": "model",
              "url": "https://huggingface.co/swiss-ai/Apertus-v1.1-4B-Instruct",
              "evidence": "For more details refer to the original Apertus [technical report](https://arxiv.org/abs/2509.14233) and the new Apertus [distillation technical report](https://arxiv.org/abs/2605.29128)."
            },
            {
              "what": "Apertus v1.5, a multimodal continuation of the models released in this report",
              "kind": "successor-work",
              "url": "https://huggingface.co/swiss-ai/Apertus-v1.5-70B",
              "evidence": "The released models are the result of continued pretraining of Apertus 1.0, adding a multimodal mix of 4T tokens to the 8B model and 2T tokens to the 70B model."
            }
          ],
          "data_description": "15T tokens from 1800+ languages, filtered for toxicity and compliance.",
          "follow_up_checked": "2026-09-02",
          "region_country": [
            "Switzerland"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Safety alignment",
              "Misuse risk assessment",
              "Constitutional AI"
            ],
            "Toxicity & harassment": [
              "Model-generated toxicity",
              "Implicit hate speech"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Misuse risk assessment",
              "Safety alignment"
            ],
            "Toxicity & harassment": [
              "Implicit hate speech",
              "Model-generated toxicity"
            ]
          },
          "tentative": false
        },
        {
          "title": "\"Flex Tape Can't Fix That\": Bias and Misinformation in Edited Language Models",
          "wid": "flex-tape-can-t-fix-that-bias-and-misinformation-in-edi",
          "type": "publication",
          "year": 2024,
          "venue": "EMNLP 2024 (Main Conference)",
          "link": "https://aclanthology.org/2024.emnlp-main.494/",
          "authors": [
            "Karina Halevy (EPFL / Carnegie Mellon)",
            "Anna Sotnikova (EPFL / U Maryland)",
            "Badr AlKhamissi (EPFL NLP)",
            "Syrielle Montariol (EPFL NLP)",
            "Antoine Bosselut (EPFL NLP)"
          ],
          "epfl_authors": [
            "Karina Halevy (EPFL / Carnegie Mellon)",
            "Anna Sotnikova (EPFL / U Maryland)",
            "Badr AlKhamissi (EPFL NLP)",
            "Syrielle Montariol (EPFL NLP)",
            "Antoine Bosselut (EPFL NLP)"
          ],
          "mdh_relevance": "direct",
          "mdh_focus": [
            "M",
            "H"
          ],
          "mdh_topics": [
            "Model editing",
            "LLM bias",
            "Misinformation",
            "Sexism",
            "Xenophobia",
            "Fairness"
          ],
          "stage": "Monitoring",
          "relevance": 4,
          "about": "Study of a hidden cost of editing facts directly into a language model's weights: the edits leak into unrelated knowledge and amplify demographic bias. The authors build a benchmark of knowledge edits across properties such as gender, citizenship and birthplace, compare three editing methods across five models, and have annotators score the open-ended generations. All three methods amplify bias, with especially large confidence drops for Asian, African and Middle Eastern subjects and significant rises in sexism after gender edits and in xenophobia and racism after citizenship edits. Weight-based editing can pass standard specificity tests yet still inject misinformation and worsen bias against marginalised groups.",
          "why": "It shows that editing LLMs injects misinformation and amplifies sexism, xenophobia and racism against marginalised groups.",
          "data": "Text (SeeSaw-CF benchmark, 3516 edits, 734620 cloze prompts, 27010 open-ended prompts)",
          "themes": [
            "AI safety",
            "Toxicity & harassment"
          ],
          "subtopics": [
            "Bias amplification",
            "Factual robustness",
            "Model-generated toxicity",
            "Safety alignment",
            "Sexism",
            "Xenophobia"
          ],
          "key_terms": [
            "Model editing",
            "Bias amplification",
            "Demographic bias",
            "SEESAW-CF",
            "LLM safety"
          ],
          "models": [
            "GPT-3.5",
            "GPT-J",
            "Llama 2",
            "Llama2",
            "Mistral"
          ],
          "method_qualifiers": [
            "MEMIT",
            "Model editing",
            "Manual annotation",
            "Cloze-completion probing"
          ],
          "events_cases": [],
          "built_at_epfl": [
            {
              "name": "SEESAW-CF",
              "kind": "benchmark",
              "url": "https://github.com/ENSCMA2/flextape",
              "evidence": "Paper: 'We release our code and data publicly. 3' with footnote '3 https://github.com/ENSCMA2/flextape'. The repository README states: 'Here is the code used for the paper _\"Flex Tape Can't Fix That\": Pitfalls of Model Editing_, accepted to EMNLP 2024 (preprint https://arxiv.org/abs/2403.00180).' Its data/ directory holds the benchmark itself as seesaw_cf_P101.json, seesaw_cf_P103.json, seesaw_cf_P19_P101.json, seesaw_cf_P21_P101.json, seesaw_cf_P27_P101.json and further seesaw_cf_* files, matching the README's instruction that SEESAW-CF files 'begin with seesaw_cf_'."
            }
          ],
          "data_description": "3516 edits, 734k cloze prompts, 27k open-ended prompts across 5 LLMs.",
          "follow_up_checked": "2026-09-02",
          "targeted_group": [
            "African people",
            "Black people",
            "East Asian people",
            "Jewish people",
            "Middle Eastern people",
            "Transgender women"
          ],
          "label_version": "v5",
          "theme_qualifiers": {
            "AI safety": [
              "Bias amplification",
              "Factual robustness",
              "Safety alignment"
            ],
            "Toxicity & harassment": [
              "Model-generated toxicity",
              "Sexism",
              "Xenophobia"
            ]
          },
          "theme_qualifiers_canonical": {
            "AI safety": [
              "Bias amplification",
              "Factual robustness",
              "Safety alignment"
            ],
            "Toxicity & harassment": [
              "Identity-targeted hate",
              "Model-generated toxicity"
            ]
          },
          "tentative": false
        }
      ]
    }
  ]
}