diff --git a/.github/workflows/jekyll-deploy.yml b/.github/workflows/jekyll-deploy.yml index 9c2bc1917..e3d36f67b 100644 --- a/.github/workflows/jekyll-deploy.yml +++ b/.github/workflows/jekyll-deploy.yml @@ -6,6 +6,11 @@ on: push: branches: ["main", "deploy", "gh-pages"] + # Weekly rebuild so STAMINA talks move from "Upcoming" to "Past" without a push + # (Wednesdays 12:00 UTC, the day after the Tuesday talks). + schedule: + - cron: "0 12 * * 3" + # Allows you to run this workflow manually from the Actions tab workflow_dispatch: diff --git a/_config.yml b/_config.yml index db1a55f5e..fed5d2c99 100644 --- a/_config.yml +++ b/_config.yml @@ -51,6 +51,7 @@ exclude: - records - src - tests + - _stamina_talks/README.md # Plugins (previously gems:) plugins: @@ -186,6 +187,8 @@ collections: permalink: /:path/ tech_transfer: output: false + stamina_talks: # one Markdown file per STAMINA talk, rendered on /stamina/ + output: false # Performance compress_html: diff --git a/_includes/stamina-talk.html b/_includes/stamina-talk.html new file mode 100644 index 000000000..a629559c3 --- /dev/null +++ b/_includes/stamina-talk.html @@ -0,0 +1,52 @@ +{% comment %} + Renders one STAMINA talk from the _stamina_talks collection. + Usage: {% include stamina-talk.html talk=talk %} +{% endcomment %} +{%- assign talk = include.talk -%} +{%- assign talk_id = talk.date | date: "%Y%m%d" | append: "-" | append: talk.slug | slugify -%} +{%- assign title_url = talk.title_url | default: talk.links.paper -%} +{%- assign abstract = talk.content | strip -%} +

{{ talk.date | date: "%Y/%m/%d" }}

+
  • + {% if title_url %}{{ talk.title }}{% else %}{{ talk.title }}{% endif %} +
    + {{ talk.role | default: "Presenter" }}: + {%- for p in talk.presenters -%} + {%- if p.url %}{{ p.name }}{% else %}{{ p.name }}{% endif -%} + {%- unless forloop.last %}, {% endunless -%} + {%- endfor -%} + {% if talk.affiliation %}, {{ talk.affiliation | markdownify | remove: "

    " | remove: "

    " | strip }}{% endif %} + {%- if talk.bio %} + +
    +
    + {{ talk.bio | markdownify }} +
    +
    + {%- endif %} +
    + {%- if talk.links.recording %} + + {%- endif %} + {%- if talk.links.paper %} + + {%- endif %} + {%- if talk.links.code %} + + {%- endif %} + {%- if talk.links.slides %} + + {%- endif %} + {%- if abstract != "" %} + +
    +
    + {{ abstract | markdownify }} +
    +
    + {%- endif %} +
  • diff --git a/_stamina_talks/2026-02-17-orlando.md b/_stamina_talks/2026-02-17-orlando.md new file mode 100644 index 000000000..3c4aa389f --- /dev/null +++ b/_stamina_talks/2026-02-17-orlando.md @@ -0,0 +1,19 @@ +--- +date: 2026-02-17 +title: 'Emergent Coordinated Behaviors in Networked LLM Agents: Modeling the Strategic Dynamics of Information Operations' +presenters: +- name: Gian Marco Orlando +- name: Jinyi Ye +- name: Mahdi Saeedi +affiliation: University of Naples Federico II - University of Southern California (ISI) +links: + recording: https://www.youtube.com/watch?v=BH9_gyYyvdw + paper: https://arxiv.org/abs/2510.25003 +bio: |- + Gian Marco Orlando is currently a PhD Student at the University of Naples Federico II. He earned his Master’s Degree in Computer Engineering from the University of Naples Federico II, graduating with honors. His thesis highlights his expertise in the intersection of Artificial Intelligence and Social Network Analysis. His research interests lie in Social Network Analysis, Agent-Based Modeling and Big Data Analytics. + + Jinyi is a second-year CS PhD student co-advised by Dr. Emilio Ferrara and Dr. Luca Luceri. Her research lies in the intersection of computer science and social science, recently focusing on large-scale agentic simulations of human behavior, measuring and modeling collective dynamics in social networks, and empirical studies on AI and the future of work. + + Mahdi Saeedi is currently exploring large-scale LLM simulations because he is fascinated by understanding how these models actually work under the hood. His background spans both the theoretical foundations and hands-on engineering of generative AI systems, which has been great preparation for this research. He have also spent time thinking about how people interact with AI through thoughtful interface design, since he believes making these tools intuitive and accessible is just as important as the underlying technology. +--- +Generative agents are rapidly advancing in sophistication, raising urgent questions about how they might coordinate when deployed in online ecosystems. This is particularly consequential in information operations (IOs), influence campaigns that aim to manipulate public opinion on social media. While traditional IOs have been orchestrated by human operators and relied on manually crafted tactics, agentic AI promises to make campaigns more automated, adaptive, and difficult to detect. This work presents the first systematic study of emergent coordination among generative agents in simulated IO campaigns. Using generative agent-based modeling, we instantiate IO and organic agents in a simulated environment and evaluate coordination across operational regimes, from simple goal alignment to team knowledge and collective decision-making. As operational regimes become more structured, IO networks become denser and more clustered, interactions more reciprocal and positive, narratives more homogeneous, amplification more synchronized, and hashtag adoption faster and more sustained. Remarkably, simply revealing to agents which other agents share their goals can produce coordination levels nearly equivalent to those achieved through explicit deliberation and collective voting. Overall, we show that generative agents, even without human guidance, can reproduce coordination strategies characteristic of real-world IOs, underscoring the societal risks posed by increasingly automated, self-organizing IOs. diff --git a/_stamina_talks/2026-03-03-weiss.md b/_stamina_talks/2026-03-03-weiss.md new file mode 100644 index 000000000..ba9c6eb1c --- /dev/null +++ b/_stamina_talks/2026-03-03-weiss.md @@ -0,0 +1,12 @@ +--- +date: 2026-03-03 +title: AI and the Future of Science +presenters: +- name: Martin Weiss + url: https://martincsweiss.com/ +affiliation: '[Tiptree Systems](https://tiptreesystems.com/)' +links: + recording: https://www.youtube.com/watch?v=vwbpX-585qI&feature=youtu.be +bio: Martin Weiss is Co-Founder of Tiptree Systems, a startup building AI agents that help ML researchers find, create, and share knowledge more efficiently. Tiptree is deployed to researchers across many top-tier institutes including Mila, ELLIS, MIT, and many more. Martin holds a PhD in AI from Mila, where he studied under Hugo Larochelle and Chris Pal. Before his PhD, he was an early employee at YesGraph, a social graph startup acquired by Lyft. +--- +This talk examines three converging crises. First, the decoupling of control from comprehension — we can increasingly predict and manipulate systems without understanding why they work. Second, the collapse of the generator-verifier gap — AI makes it trivial to produce the aesthetics of deep thought. This makes peer review more difficult because we can no longer rely on easy-to-verify signals of work quality. Third, the credit assignment gap — our academic reward systems optimize for publication metrics, not the increase in understanding that a new paper produces. diff --git a/_stamina_talks/2026-03-10-faulkner.md b/_stamina_talks/2026-03-10-faulkner.md new file mode 100644 index 000000000..391eddf02 --- /dev/null +++ b/_stamina_talks/2026-03-10-faulkner.md @@ -0,0 +1,14 @@ +--- +date: 2026-03-10 +title: Evaluating Cooperation in LLM Social Groups through Self-Organizing Leadership +title_url: https://drive.google.com/file/d/1UNVlGqzhnh2BNpviwctUlvue33MjN34k/view +presenters: +- name: Ryan Faulkner + url: https://www.cs.toronto.edu/~rfaulk/ +affiliation: University of Toronto/Deepmind +links: + recording: https://youtu.be/NaNEwyxjeXo + paper: https://drive.google.com/file/d/1UNVlGqzhnh2BNpviwctUlvue33MjN34k/view?usp=drive_link +bio: Ryan is a Computer Scientist and Machine Learning researcher with a background in reinforcement learning and foundation models. He has worked as a Research Engineer over the past decade at Google Deepmind and he is also a PhD Student at the University of Toronto advised by Zhijing Jin. At GDM he works in the Concordia group led by Joel Leibo. At a high level his current research focus is on multi-agent systems, LLMs, and social learning. In this context he is interested in memory mechanisms, agent theory of mind, collective decision making, and simulating political systems. +--- +Governing common-pool resources requires agents to develop enduring strategies through cooperation and self-governance to avoid collective failure. While foundation models have shown potential for cooperation in these settings, existing multi-agent research provides little insight into whether structured leadership and election mechanisms can improve collective decision making. The lack of such a critical organizational feature ubiquitous in human society presents a significant shortcoming of the current methods. In this work we aim to directly address whether leadership and elections can support improved social welfare and cooperation through multi-agent simulation with LLMs. We present a new framework that simulates leadership through elected personas and candidate-driven agendas and carry out an empirical study of LLMs under controlled governance conditions. Our experiments demonstrate that structured leadership can improve social welfare scores by 55.4% and survival time by 128.6% across a range of high performing LLMs. Through the construction of an agent social graph we compute centrality metrics to assess the social influence of leader personas and also analyze rhetorical and cooperative tendencies revealed through a sentiment analysis on leader utterances. This work lays the foundation for developing prosocial, self-governing multi-agent systems capable of navigating complex resource dilemmas. diff --git a/_stamina_talks/2026-03-24-parent.md b/_stamina_talks/2026-03-24-parent.md new file mode 100644 index 000000000..b0190ce94 --- /dev/null +++ b/_stamina_talks/2026-03-24-parent.md @@ -0,0 +1,21 @@ +--- +date: 2026-03-24 +title: AI and the knowledge commons +title_url: https://www.conversence.com/presentations/2026-03-24-stamina.pdf +presenters: +- name: Marc-Antoine Parent + url: https://conversence.com +affiliation: Solutions Conversence inc. +links: + recording: https://www.youtube.com/watch?v=Q4WcKmfOtx4 + slides: https://www.conversence.com/presentations/2026-03-24-stamina.pdf +bio: |- + Marc-Antoine has worked in computational linguistics, knowledge representation, and more recently has focused on tools for augmented collective intelligence. He's especially interested in how to represent emergent and disputed knowledge. + + [https://conversence.com](https://conversence.com) + + [https://hyperknowledge.org](https://hyperknowledge.org) +--- +We present a model of the formation of a knowledge commons for democratic, collective decision-making in society, and explain how generative AI disrupts the formation of this knowledge commons. We will also present ways in which to reinforce the collective processes around a knowledge commons, including the possible contributions of hybrid AI. + +This is based on a paper that was presented at the IJCAI'25 democracy and AI workshop. diff --git a/_stamina_talks/2026-04-14-jin.md b/_stamina_talks/2026-04-14-jin.md new file mode 100644 index 000000000..ed67fec50 --- /dev/null +++ b/_stamina_talks/2026-04-14-jin.md @@ -0,0 +1,16 @@ +--- +date: 2026-04-14 +title: Testing and Improving Multi-Agent LLM Cooperation +presenters: +- name: Zhijing Jin + url: https://zhijing-jin.com/ +affiliation: University of Toronto +links: + recording: https://www.youtube.com/watch?v=Bme6Q8nKfrs +bio: Zhijing Jin (she/her) is an Assistant Professor at the University of Toronto and Research Scientist at the Max Planck Institute. She serves as a CIFAR AI Chair, an ELLIS advisor, and a faculty member at the Vector Institute, and the Schwartz Reisman Institute. She co-chairs the ACL Ethics Committee, and the ACL Year-Round Mentorship. Her research focuses on Causal Reasoning with LLMs, and AI Safety in Multi-Agent LLMs. She has published over 80 papers and has received the ELLIS PhD Award, three Rising Star awards, and two Best Paper awards at NeurIPS 2024 Workshops. +--- +While progress has been made in evaluating single-agent LLMs for persona modeling, the behavior of these models within multi-agent groups remains underexplored. This presentation outlines a research series dedicated to closing this gap by testing LLM cooperation through autonomous social simulations. Specifically, we ask: what happens when personas are tasked to interact and cooperate? + +To answer this, we introduce a suite of simulation environments (GovSim, MoralSim, and SanctSim) designed to stress-test persona interaction. These environments simulate high-stakes scenarios, such as the tragedy of the commons and ethical trade-offs, allowing us to investigate whether simulated societies can autonomously negotiate social order and how personas with differing ethical constraints navigate social dilemmas. + +Our findings highlight implications for persona modeling. We show that agents exhibit a functional "theory of mind," capable of inferring the identities of their interlocutors and strategically adapting their behavior, sometimes exploiting specific model vulnerabilities. Furthermore, we discuss a counterintuitive phenomenon where advanced reasoning capabilities lead to exploitative behaviors that humans typically avoid, highlighting a significant misalignment between agent optimization and human social norms. diff --git a/_stamina_talks/2026-04-28-tamari.md b/_stamina_talks/2026-04-28-tamari.md new file mode 100644 index 000000000..d36e4db02 --- /dev/null +++ b/_stamina_talks/2026-04-28-tamari.md @@ -0,0 +1,12 @@ +--- +date: 2026-04-28 +title: From Social Networks to Sensemaking Networks +presenters: +- name: Ronen Tamari + url: https://cosmik.network/ +affiliation: Cosmik Network +links: + recording: https://youtu.be/Ew0Co8hN98U +bio: Ronen is a researcher and entrepreneur working on collective intelligence systems to help us think better, together. Ronen recently completed an Open Science fellowship at the Astera Institute, where he co-founded Cosmik, a mission driven R&D lab working on new kinds of social networks for collective sensemaking. Ronen also completed a PhD in computer science, with a focus on cognitive-inspired AI models for natural language comprehension. Ronen’s current research interests center around cooperative human-AI systems, institutional design for collective intelligence, and the role of epistemic environments in shaping human and machine intelligence. +--- +What would social media look like if it were designed for sensemaking rather than engagement? We're exploring this question with Semble, a platform where researchers curate shareable collections, create knowledge trails that others can build on, and discover relevant work through their network's collective attention. Built on the AT Protocol, the open social networking protocol behind Bluesky, Semble offers researchers data portability and an open API designed for extension. We'll discuss how Semble enables new kinds of research tooling, from living semantic citation graphs to collaborative review and annotation. We'll also share how ATProto's open data layer creates unique opportunities for studying and designing epistemic infrastructure — from observing how knowledge trails form across a network to experimenting with platform affordances that support collective sensemaking. diff --git a/_stamina_talks/2026-05-05-abdulhai.md b/_stamina_talks/2026-05-05-abdulhai.md new file mode 100644 index 000000000..4237453e0 --- /dev/null +++ b/_stamina_talks/2026-05-05-abdulhai.md @@ -0,0 +1,12 @@ +--- +date: 2026-05-05 +title: Consistently Simulating Human Personas with Multi-Turn Reinforcement Learning +presenters: +- name: Marwa Abdulhai + url: https://abdulhaim.github.io/ +affiliation: UC Berkeley AI Research (BAIR) Lab +links: + recording: https://youtu.be/4sA8Xe6mCZQ +bio: Marwa Abdulhai is a PhD candidate at UC Berkeley advised by Sergey Levine. Her research focuses on enabling AI agents to better understand people and their interactions to build both safe and more AI capable systems. This includes improving the performance of existing large language models (LLMs) for multi-turn dialogue interactions, understanding how to protect against deception in AI systems, and exploring how AI can serve as a useful tool for social science research. Her research has been supported by the Quad Fellowship, AI Policy Hub, Open AI Research, and Cooperative AI PhD Fellowship. +--- +Large Language Models (LLMs) are increasingly used to simulate human users in interactive settings such as therapy, education, and social role-play. While these simulations enable scalable training and evaluation of AI agents, off-the-shelf LLMs often drift from their assigned personas, contradict earlier statements, or abandon role-appropriate behavior. We introduce a unified framework for evaluating and improving consistency in LLM-generated dialogue with multi-turn RL, reducing inconsistency by over 55%, resulting in more coherent and trustworthy simulated users. diff --git a/_stamina_talks/2026-05-19-xiao.md b/_stamina_talks/2026-05-19-xiao.md new file mode 100644 index 000000000..7adf99f49 --- /dev/null +++ b/_stamina_talks/2026-05-19-xiao.md @@ -0,0 +1,12 @@ +--- +date: 2026-05-19 +title: 'The Chameleon''s Limit: Investigating Persona Collapse and Homogenization in Large Language Models' +presenters: +- name: Yunze (Lorenzo) Xiao + url: https://algoroxyolo.github.io/ +affiliation: CMU Technologies Institute (LTI) +links: + recording: https://youtu.be/I06zCefkkdg +bio: Yunze (Lorenzo) Xiao is a Master’s student at Carnegie Mellon University’s Language Technologies Institute, advised by Prof. Mona Diab. His research aims to develop large language models that move beyond surface-level fluency toward genuine human-like intelligence, spanning anthropomorphism as a controllable modeling dimension, persona consistency, long-horizon memory, affective simulation, and multi-agent systems, with applications in education and therapy. He has published at ACL, EMNLP, and LREC-COLING, including InCharacter and ToxiCloakCN, and previously conducted research at SUTD and QCRI on multilingual NLP, propaganda detection, and AI safety. He co-organized the NeurIPS 2025 PersonaLLM workshop. +--- +Applications based on large language models (LLMs), such as multi-agent simulations, require population diversity among agents. We identify a pervasive failure mode we term *Persona Collapse*: agents each assigned a distinct profile nonetheless converge into a narrow behavioral mode, producing a homogeneous simulated population. To quantify persona collapse, we propose a framework that measures how much of the persona space a population occupies (Coverage), how evenly agents spread across it (Uniformity), and how rich the resulting behavioral patterns are (Complexity). Evaluating ten LLMs on personality simulation (BFI-44), moral reasoning, and self-introduction, we observe persona collapse along two axes: (1) Dimensions: a model can appear diverse on one axis yet structurally degenerate on another, and (2) Domains: the same model may collapse the most in personality yet be the most diverse in moral reasoning. Furthermore, item-level diagnostics reveal that behavioral variation tracks coarse demographic stereotypes rather than the fine-grained individual differences specified in each persona. Counter-intuitively, **the models achieving the highest per-persona fidelity consistently produce the most stereotyped populations**. We release our toolkit and data to support population-level evaluation of LLMs. diff --git a/_stamina_talks/2026-06-02-stengel-eskin.md b/_stamina_talks/2026-06-02-stengel-eskin.md new file mode 100644 index 000000000..f09f74350 --- /dev/null +++ b/_stamina_talks/2026-06-02-stengel-eskin.md @@ -0,0 +1,12 @@ +--- +date: 2026-06-02 +title: Multi-Model Training for Multi-Agent Communication Skills +presenters: +- name: Elias Stengel-Eskin + url: https://esteng.github.io +affiliation: University of Texas at Austin +links: + recording: https://youtu.be/ETLD0Yr0bkM +bio: Elias Stengel-Eskin is an Assistant Professor of Computer Science at the University of Texas at Austin. His research spans natural language processing, computational linguistics, and artificial intelligence, and focuses on developing AI agents that can intelligently communicate and collaborate with people and each other. This includes work on multi-agent communication and collaboration, converting language to action, grounding language to vision, and handling uncertainty, ambiguity, and underspecification. Before joining UT Austin, he was a Postdoctoral Research Associate at the University of North Carolina, Chapel Hill. He received a Ph.D. in Computer Science in 2023 from Johns Hopkins University and a B.A. & Sc. in Cognitive Science from McGill University in 2018. +--- +As we scale from individual agents to teams of agents, inter-agent communication will become increasingly important. In this talk, I will describe a general paradigm for teaching multi-agent communication skills through multi-model reinforcement learning, which I will illustrate via three key collaborative skills: expressing confidence in a calibrated way, responding robustly to positive and negative persuasion, and expressing reasoning faithfully. I will show how these problems can be framed in terms of speaker-listener games, and how this framing allows us to teach models collaborative skills, often using games simulated on smaller models to train larger models. diff --git a/_stamina_talks/2026-06-23-stamina-matters.md b/_stamina_talks/2026-06-23-stamina-matters.md new file mode 100644 index 000000000..b19bdf064 --- /dev/null +++ b/_stamina_talks/2026-06-23-stamina-matters.md @@ -0,0 +1,10 @@ +--- +date: 2026-06-23 +title: '1st STAMINA Matters informational meeting: An Overview and Discussion of STAMINA Working Group Activities and Plans' +presenters: +- name: Maximilian Puelma Touzel +role: Facilitator +links: + recording: https://youtu.be/rGPOAAI75IA +--- +After 6 months of stimulating STAMINA talks at the intersection of social tech, social science, and AI safety, we are organizing our 1st informational meeting! We'll give overview of emerging STAMINA activities (our fresh slack workspace, upcoming conference workshops, shared tasks etc.) and discuss what the STAMINA interested community members are looking for. Looking forward to seeing you there! diff --git a/_stamina_talks/README.md b/_stamina_talks/README.md new file mode 100644 index 000000000..884b3d8ef --- /dev/null +++ b/_stamina_talks/README.md @@ -0,0 +1,23 @@ +# STAMINA talks + +Each file in this folder is one talk on the [STAMINA page](../stamina/index.html). +The page is built from these files by Jekyll (`_includes/stamina-talk.html` renders one talk). + +**Add a talk:** copy `_template.md` to `YYYY-MM-DD-lastname.md`, fill in the fields, and commit. + +**After the talk:** add the YouTube link under `links: recording:`. + +A talk is listed under *Upcoming* until the day after its date, then under *Past*, +grouped by term (Jan–Jun = Spring, Jul–Dec = Fall; override with `term:`). +The site rebuilds on every push to `main` and every Wednesday, so talks move to *Past* on their own. + +**Preview locally** (from the repo root, before pushing): + +```bash +bundle exec jekyll build # ~35 s; writes the site to _site/ +python3 -m http.server 4000 -d _site +# open http://localhost:4000/stamina/ ; re-run the build after each edit, then refresh +``` + +(`jekyll serve` / `build --watch` crash on WSL with Ruby 3.0 + Jekyll 3.9 — +"no implicit conversion of Hash into Integer" in pathutil — so use the two commands above.) diff --git a/_stamina_talks/_template.md b/_stamina_talks/_template.md new file mode 100644 index 000000000..4c73003f6 --- /dev/null +++ b/_stamina_talks/_template.md @@ -0,0 +1,22 @@ +--- +# Copy this file to YYYY-MM-DD-speakerlastname.md (e.g. 2026-10-06-smith.md). +# Files starting with "_" (like this one) are ignored by the site. +# Delete any optional field you don't need; empty badges are not shown. +date: 2026-10-06 +title: "Talk title" +title_url: # optional; the title links to links.paper if this is empty +presenters: # one or more + - name: Presenter Name + url: https://example.com # optional +# role: Facilitator # optional; defaults to "Presenter" +affiliation: University Name # Markdown allowed, e.g. "[Lab](https://lab.org)" +# term: "Fall 2026" # optional; overrides the automatic Spring/Fall heading +links: + recording: # YouTube link, add after the talk + paper: + code: + slides: +bio: | + Speaker bio in Markdown. Leave a blank line between paragraphs. +--- +Abstract goes here, in Markdown. Leave a blank line between paragraphs. diff --git a/stamina/index.html b/stamina/index.html index 7b476b25e..559c8523f 100644 --- a/stamina/index.html +++ b/stamina/index.html @@ -1,3 +1,6 @@ +--- +layout: null # standalone Bootstrap page; don't wrap in the site theme +--- @@ -144,6 +147,16 @@

    About STAMINA

    + + {%- assign today = site.time | date: "%Y%m%d" | plus: 0 -%} + {%- assign all_talks = site.stamina_talks | sort: "date" -%} + @@ -155,42 +168,20 @@

    About STAMINA

    Upcoming Talks

    + {% assign n_upcoming = 0 %} + {%- for talk in all_talks -%} + {%- assign talk_day = talk.date | date: "%Y%m%d" | plus: 0 -%} + {%- if talk_day >= today -%} + {%- assign n_upcoming = n_upcoming | plus: 1 %} +{% include stamina-talk.html talk=talk %} + {%- endif -%} + {%- endfor %} + {%- if n_upcoming == 0 %} +

    The schedule for Fall 2026 will be announced soon.

    + {%- endif %} - -
    - +
    @@ -199,8 +190,9 @@

    [DATE Y/M/D]

    @@ -210,310 +202,27 @@

    [DATE Y/M/D]

    Past Talks (Youtube Playlist)

    - -

    Spring 2026

    - -

    2026/06/23

    -
  • - 1st STAMINA Matters informational meeting: An Overview and Discussion of STAMINA Working Group Activities and Plans -
    - Facilitator: Maximilian Puelma Touzel -
    - - - - --> - -
    -
    - After 6 months of stimulating STAMINA talks at the intersection of social tech, social science, and AI safety, we are organizing our 1st informational meeting! -We'll give overview of emerging STAMINA activities (our fresh slack workspace, upcoming conference workshops, shared tasks etc.) and discuss what the STAMINA interested community members are looking for. Looking forward to seeing you there! -
    -
    -
  • - -

    2026/06/02

    -
  • - Multi-Model Training for Multi-Agent Communication Skills -
    - Presenter: Elias Stengel-Eskin, University of Texas at Austin - -
    -
    - Elias Stengel-Eskin is an Assistant Professor of Computer Science at the University of Texas at Austin. His research spans natural language processing, computational linguistics, and artificial intelligence, and focuses on developing AI agents that can intelligently communicate and collaborate with people and each other. This includes work on multi-agent communication and collaboration, converting language to action, grounding language to vision, and handling uncertainty, ambiguity, and underspecification. Before joining UT Austin, he was a Postdoctoral Research Associate at the University of North Carolina, Chapel Hill. He received a Ph.D. in Computer Science in 2023 from Johns Hopkins University and a B.A. & Sc. in Cognitive Science from McGill University in 2018. -
    -
    -
    - - - - - -
    -
    - As we scale from individual agents to teams of agents, inter-agent communication will become increasingly important. In this talk, I will describe a general paradigm for teaching multi-agent communication skills through multi-model reinforcement learning, which I will illustrate via three key collaborative skills: expressing confidence in a calibrated way, responding robustly to positive and negative persuasion, and expressing reasoning faithfully. I will show how these problems can be framed in terms of speaker-listener games, and how this framing allows us to teach models collaborative skills, often using games simulated on smaller models to train larger models. -
    -
    -
  • - - - - -

    2026/05/19

    -
  • - The Chameleon's Limit: Investigating Persona Collapse and Homogenization in Large Language Models -
    - Presenter: Yunze (Lorenzo) Xiao, CMU Technologies Institute (LTI) - -
    -
    - Yunze (Lorenzo) Xiao is a Master’s student at Carnegie Mellon University’s Language Technologies Institute, advised by Prof. Mona Diab. His research aims to develop large language models that move beyond surface-level fluency toward genuine human-like intelligence, spanning anthropomorphism as a controllable modeling dimension, persona consistency, long-horizon memory, affective simulation, and multi-agent systems, with applications in education and therapy. He has published at ACL, EMNLP, and LREC-COLING, including InCharacter and ToxiCloakCN, and previously conducted research at SUTD and QCRI on multilingual NLP, propaganda detection, and AI safety. He co-organized the NeurIPS 2025 PersonaLLM workshop. -
    -
    -
    - - - - - -
    -
    - Applications based on large language models (LLMs), such as multi-agent simulations, require population diversity among agents. We identify a pervasive failure mode we term \emph{Persona Collapse}: agents each assigned a distinct profile nonetheless converge into a narrow behavioral mode, producing a homogeneous simulated population. To quantify persona collapse, we propose a framework that measures how much of the persona space a population occupies (Coverage), how evenly agents spread across it (Uniformity), and how rich the resulting behavioral patterns are (Complexity). Evaluating ten LLMs on personality simulation (BFI-44), moral reasoning, and self-introduction, we observe persona collapse along two axes: (1) Dimensions: a model can appear diverse on one axis yet structurally degenerate on another, and (2) Domains: the same model may collapse the most in personality yet be the most diverse in moral reasoning. Furthermore, item-level diagnostics reveal that behavioral variation tracks coarse demographic stereotypes rather than the fine-grained individual differences specified in each persona. Counter-intuitively, \textbf{the models achieving the highest per-persona fidelity consistently produce the most stereotyped populations}. We release our toolkit and data to support population-level evaluation of LLMs. -
    -
    -
  • - - -

    2026/05/05

    -
  • - Consistently Simulating Human Personas with Multi-Turn Reinforcement Learning -
    - Presenter: Marwa Abdulhai, UC Berkeley AI Research (BAIR) Lab - -
    -
    - Marwa Abdulhai is a PhD candidate at UC Berkeley advised by Sergey Levine. Her research focuses on enabling AI agents to better understand people and their interactions to build both safe and more AI capable systems. This includes improving the performance of existing large language models (LLMs) for multi-turn dialogue interactions, understanding how to protect against deception in AI systems, and exploring how AI can serve as a useful tool for social science research. Her research has been supported by the Quad Fellowship, AI Policy Hub, Open AI Research, and Cooperative AI PhD Fellowship. -
    -
    -
    - - - - - -
    -
    - Large Language Models (LLMs) are increasingly used to simulate human users in interactive settings such as therapy, education, and social role-play. While these simulations enable scalable training and evaluation of AI agents, off-the-shelf LLMs often drift from their assigned personas, contradict earlier statements, or abandon role-appropriate behavior. We introduce a unified framework for evaluating and improving consistency in LLM-generated dialogue with multi-turn RL, reducing inconsistency by over 55%, resulting in more coherent and trustworthy simulated users. -
    -
    -
  • - -
    - -

    2026/04/28

    -
  • - From Social Networks to Sensemaking Networks -
    - Presenter: Ronen Tamari, Cosmik Network - -
    -
    - Ronen is a researcher and entrepreneur working on collective intelligence systems to help us think better, together. Ronen recently completed an Open Science fellowship at the Astera Institute, where he co-founded Cosmik, a mission driven R&D lab working on new kinds of social networks for collective sensemaking. Ronen also completed a PhD in computer science, with a focus on cognitive-inspired AI models for natural language comprehension. Ronen’s current research interests center around cooperative human-AI systems, institutional design for collective intelligence, and the role of epistemic environments in shaping human and machine intelligence. -
    -
    -
    - - - - - -
    -
    - What would social media look like if it were designed for sensemaking rather than engagement? We're exploring this question with Semble, a platform where researchers curate shareable collections, create knowledge trails that others can build on, and discover relevant work through their network's collective attention. Built on the AT Protocol, the open social networking protocol behind Bluesky, Semble offers researchers data portability and an open API designed for extension. We'll discuss how Semble enables new kinds of research tooling, from living semantic citation graphs to collaborative review and annotation. We'll also share how ATProto's open data layer creates unique opportunities for studying and designing epistemic infrastructure — from observing how knowledge trails form across a network to experimenting with platform affordances that support collective sensemaking. -
    -
    -
  • - -

    2026/04/14

    -
  • - Testing and Improving Multi-Agent LLM Cooperation -
    - Presenter: Zhijing Jin, University of Toronto - -
    -
    - Zhijing Jin (she/her) is an Assistant Professor at the University of Toronto and Research Scientist at the Max Planck Institute. She serves as a CIFAR AI Chair, an ELLIS advisor, and a faculty member at the Vector Institute, and the Schwartz Reisman Institute. She co-chairs the ACL Ethics Committee, and the ACL Year-Round Mentorship. Her research focuses on Causal Reasoning with LLMs, and AI Safety in Multi-Agent LLMs. She has published over 80 papers and has received the ELLIS PhD Award, three Rising Star awards, and two Best Paper awards at NeurIPS 2024 Workshops. -
    -
    -
    - - - - - -
    -
    - While progress has been made in evaluating single-agent LLMs for persona modeling, the behavior of these models within multi-agent groups remains underexplored. This presentation outlines a research series dedicated to closing this gap by testing LLM cooperation through autonomous social simulations. Specifically, we ask: what happens when personas are tasked to interact and cooperate? -
    - To answer this, we introduce a suite of simulation environments (GovSim, MoralSim, and SanctSim) designed to stress-test persona interaction. These environments simulate high-stakes scenarios, such as the tragedy of the commons and ethical trade-offs, allowing us to investigate whether simulated societies can autonomously negotiate social order and how personas with differing ethical constraints navigate social dilemmas. -
    - Our findings highlight implications for persona modeling. We show that agents exhibit a functional "theory of mind," capable of inferring the identities of their interlocutors and strategically adapting their behavior, sometimes exploiting specific model vulnerabilities. Furthermore, we discuss a counterintuitive phenomenon where advanced reasoning capabilities lead to exploitative behaviors that humans typically avoid, highlighting a significant misalignment between agent optimization and human social norms. -
    -
    -
  • - -

    2026/03/24

    -
  • - AI and the knowledge commons -
    - Presenter: Marc-Antoine Parent, Solutions Conversence inc. - -
    -
    - Marc-Antoine has worked in computational linguistics, knowledge representation, and more recently has focused on tools for augmented collective intelligence. He's especially interested in how to represent emergent and disputed knowledge. -
    https://conversence.com -
    https://hyperknowledge.org -
    -
    -
    - - - - - -
    -
    - We present a model of the formation of a knowledge commons for democratic, collective decision-making in society, and explain how generative AI disrupts the formation of this knowledge commons. We will also present ways in which to reinforce the collective processes around a knowledge commons, including the possible contributions of hybrid AI. -
    - This is based on a paper that was presented at the IJCAI'25 democracy and AI workshop. -
    -
    -
  • - -
    - -

    2026/03/10

    -
  • - Evaluating Cooperation in LLM Social Groups through Self-Organizing Leadership -
    - Presenter: Ryan Faulkner, University of Toronto/Deepmind - -
    -
    - Ryan is a Computer Scientist and Machine Learning researcher with a background in reinforcement learning and foundation models. He has worked as a Research Engineer over the past decade at Google Deepmind and he is also a PhD Student at the University of Toronto advised by Zhijing Jin. At GDM he works in the Concordia group led by Joel Leibo. At a high level his current research focus is on multi-agent systems, LLMs, and social learning. In this context he is interested in memory mechanisms, agent theory of mind, collective decision making, and simulating political systems. - -
    -
    -
    - - - - - -
    -
    - Governing common-pool resources requires agents to develop enduring strategies through cooperation and self-governance to avoid collective failure. While foundation models have shown potential for cooperation in these settings, existing multi-agent research provides little insight into whether structured leadership and election mechanisms can improve collective decision making. The lack of such a critical organizational feature ubiquitous in human society presents a significant shortcoming of the current methods. In this work we aim to directly address whether leadership and elections can support improved social welfare and cooperation through multi-agent simulation with LLMs. We present a new framework that simulates leadership through elected personas and candidate-driven agendas and carry out an empirical study of LLMs under controlled governance conditions. Our experiments demonstrate that structured leadership can improve social welfare scores by 55.4% and survival time by 128.6% across a range of high performing LLMs. Through the construction of an agent social graph we compute centrality metrics to assess the social influence of leader personas and also analyze rhetorical and cooperative tendencies revealed through a sentiment analysis on leader utterances. This work lays the foundation for developing prosocial, self-governing multi-agent systems capable of navigating complex resource dilemmas. -
    -
    -
  • - -
    - -

    2026/03/03

    -
  • - AI and the Future of Science -
    - Presenter: Martin Weiss, Tiptree Systems - -
    -
    - Martin Weiss is Co-Founder of Tiptree Systems, a startup building AI agents that help ML researchers find, create, and share knowledge more efficiently. Tiptree is deployed to researchers across many top-tier institutes including Mila, ELLIS, MIT, and many more. Martin holds a PhD in AI from Mila, where he studied under Hugo Larochelle and Chris Pal. Before his PhD, he was an early employee at YesGraph, a social graph startup acquired by Lyft. -
    -
    -
    - - - - - -
    -
    - This talk examines three converging crises. First, the decoupling of control from comprehension — we can increasingly predict and manipulate systems without understanding why they work. Second, the collapse of the generator-verifier gap — AI makes it trivial to produce the aesthetics of deep thought. This makes peer review more difficult because we can no longer rely on easy-to-verify signals of work quality. Third, the credit assignment gap — our academic reward systems optimize for publication metrics, not the increase in understanding that a new paper produces. -
    -
    -
  • - -
    - -

    2026/02/17

    -
  • - Emergent Coordinated Behaviors in Networked LLM Agents: Modeling the Strategic Dynamics of Information Operations -
    - Presenter: Gian Marco Orlando, Jinyi Ye, Mahdi Saeedi, University of Naples Federico II - University of Southern California (ISI) - -
    -
    - Gian Marco Orlando is currently a PhD Student at the University of Naples Federico II. He earned his Master’s Degree in Computer Engineering from the University of Naples Federico II, graduating with honors. His thesis highlights his expertise in the intersection of Artificial Intelligence and Social Network Analysis. His research interests lie in Social Network Analysis, Agent-Based Modeling and Big Data Analytics. -

    - Jinyi is a second-year CS PhD student co-advised by Dr. Emilio Ferrara and Dr. Luca Luceri. Her research lies in the intersection of computer science and social science, recently focusing on large-scale agentic simulations of human behavior, measuring and modeling collective dynamics in social networks, and empirical studies on AI and the future of work. -

    - Mahdi Saeedi is currently exploring large-scale LLM simulations because he is fascinated by understanding how these models actually work under the hood. His background spans both the theoretical foundations and hands-on engineering of generative AI systems, which has been great preparation for this research. He have also spent time thinking about how people interact with AI through thoughtful interface design, since he believes making these tools intuitive and accessible is just as important as the underlying technology. -
    -
    -
    - - - - - -
    -
    - Generative agents are rapidly advancing in sophistication, raising urgent questions about how they might coordinate when deployed in online ecosystems. This is particularly consequential in information operations (IOs), influence campaigns that aim to manipulate public opinion on social media. While traditional IOs have been orchestrated by human operators and relied on manually crafted tactics, agentic AI promises to make campaigns more automated, adaptive, and difficult to detect. This work presents the first systematic study of emergent coordination among generative agents in simulated IO campaigns. Using generative agent-based modeling, we instantiate IO and organic agents in a simulated environment and evaluate coordination across operational regimes, from simple goal alignment to team knowledge and collective decision-making. As operational regimes become more structured, IO networks become denser and more clustered, interactions more reciprocal and positive, narratives more homogeneous, amplification more synchronized, and hashtag adoption faster and more sustained. - Remarkably, simply revealing to agents which other agents share their goals can produce coordination levels nearly equivalent to those achieved through explicit deliberation and collective voting. - Overall, we show that generative agents, even without human guidance, can reproduce coordination strategies characteristic of real-world IOs, underscoring the societal risks posed by increasingly automated, self-organizing IOs. -
    -
    -
  • - + {% assign current_term = "" %} + {%- assign past_talks = all_talks | reverse -%} + {%- for talk in past_talks -%} + {%- assign talk_day = talk.date | date: "%Y%m%d" | plus: 0 -%} + {%- if talk_day < today -%} + {%- assign talk_month = talk.date | date: "%-m" | plus: 0 -%} + {%- if talk.term -%} + {%- assign talk_term = talk.term -%} + {%- elsif talk_month <= 6 -%} + {%- assign talk_term = talk.date | date: "%Y" | prepend: "Spring " -%} + {%- else -%} + {%- assign talk_term = talk.date | date: "%Y" | prepend: "Fall " -%} + {%- endif -%} + {%- if talk_term != current_term %} +

    {{ talk_term }}

    + {%- assign current_term = talk_term -%} + {%- endif %} +{% include stamina-talk.html talk=talk %}
    + {%- endif -%} + {%- endfor %}
    diff --git a/tests/data/add_publication_by_id/2020-08-01-2004.09456.md b/tests/data/add_publication_by_id/2020-08-01-2004.09456.md index 82caaa388..c46681e87 100644 --- a/tests/data/add_publication_by_id/2020-08-01-2004.09456.md +++ b/tests/data/add_publication_by_id/2020-08-01-2004.09456.md @@ -3,7 +3,7 @@ title: 'StereoSet: Measuring stereotypical bias in pretrained language models' venue: Annual Meeting of the Association for Computational Linguistics openAccessPdf: url: https://aclanthology.org/2021.acl-long.416.pdf - status: HYBRID + status: GOLD license: CCBY disclaimer: 'Notice: Paper or abstract available at https://arxiv.org/abs/2004.09456, which is subject to the license by the author or copyright owner provided with