
@inproceedings{bender_dangers_2021,
	address = {Virtual Event Canada},
	title = {On the {Dangers} of {Stochastic} {Parrots}: {Can} {Language} {Models} {Be} {Too} {Big}?},
	isbn = {978-1-4503-8309-7},
	shorttitle = {On the {Dangers} of {Stochastic} {Parrots}},
	url = {https://dl.acm.org/doi/10.1145/3442188.3445922},
	doi = {10.1145/3442188.3445922},
	abstract = {The past 3 years of work in NLP have been characterized by the development and deployment of ever larger language models, especially for English. BERT, its variants, GPT-2/3, and others, most recently Switch-C, have pushed the boundaries of the possible both through architectural innovations and through sheer size. Using these pretrained models and the methodology of fine-tuning them for specific tasks, researchers have extended the state of the art on a wide array of tasks as measured by leaderboards on specific benchmarks for English. In this paper, we take a step back and ask: How big is too big? What are the possible risks associated with this technology and what paths are available for mitigating those risks? We provide recommendations including weighing the environmental and financial costs first, investing resources into curating and carefully documenting datasets rather than ingesting everything on the web, carrying out pre-development exercises evaluating how the planned approach fits into research and development goals and supports stakeholder values, and encouraging research directions beyond ever larger language models.},
	language = {en},
	urldate = {2021-05-17},
	booktitle = {Proceedings of the 2021 {ACM} {Conference} on {Fairness}, {Accountability}, and {Transparency}},
	publisher = {ACM},
	author = {Bender, Emily M. and Gebru, Timnit and McMillan-Major, Angelina and Shmitchell, Shmargaret},
	month = mar,
	year = {2021},
	pages = {610--623},
	file = {Bender et al_2021_On the Dangers of Stochastic Parrots.pdf:/Users/eddieungless/OneDrive - University of Edinburgh/Zotero Attachments/Bender et al_2021_On the Dangers of Stochastic Parrots2.pdf:application/pdf},
}

@inproceedings{markl_mind_2022,
	address = {Dublin, Ireland},
	title = {Mind the data gap(s): {Investigating} power in speech and language datasets},
	shorttitle = {Mind the data gap(s)},
	url = {https://aclanthology.org/2022.ltedi-1.1},
	doi = {10.18653/v1/2022.ltedi-1.1},
	abstract = {Algorithmic oppression is an urgent and persistent problem in speech and language technologies. Considering power relations embedded in datasets before compiling or using them to train or test speech and language technologies is essential to designing less harmful, more just technologies. This paper presents a reflective exercise to recognise and challenge gaps and the power relations they reveal in speech and language datasets by applying principles of Data Feminism and Design Justice, and building on work on dataset documentation and sociolinguistics.},
	urldate = {2022-08-22},
	booktitle = {Proceedings of the {Second} {Workshop} on {Language} {Technology} for {Equality}, {Diversity} and {Inclusion}},
	publisher = {Association for Computational Linguistics},
	author = {Markl, Nina},
	month = may,
	year = {2022},
	pages = {1--12},
	file = {Markl_2022_Mind the data gap(s).pdf:/Users/eddieungless/OneDrive - University of Edinburgh/Zotero Attachments/Markl_2022_Mind the data gap(s).pdf:application/pdf},
}

@incollection{jones_lip-synching_2023,
	title = {Lip-synching and young people's everyday linguistic activism on {TikTok}},
	url = {https://www.vr-elibrary.de/doi/abs/10.14220/9783737016391.23},
	booktitle = {\#{YouthMediaLife} \& {Friends}},
	publisher = {V\&R unipress},
	author = {Jones, Rodney H.},
	year = {2023},
	pages = {23--42},
}

@article{kachel_i_2018,
	title = {“{Do} {I} {Sound} {Straight}?”: {Acoustic} {Correlates} of {Actual} and {Perceived} {Sexual} {Orientation} and {Masculinity}/{Femininity} in {Men}'s {Speech}},
	volume = {61},
	shorttitle = {“{Do} {I} {Sound} {Straight}?},
	url = {https://pubs.asha.org/doi/full/10.1044/2018_JSLHR-S-17-0125},
	doi = {10.1044/2018_JSLHR-S-17-0125},
	abstract = {Purpose 

This study aims to give an integrative answer on which speech stereotypes exist toward German gay and straight men, whether and how acoustic correlates of actual and perceived sexual orientation are connected, and how this relates to masculinity/femininity. Hence, it tests speech stereotype accuracy in the context of sexual orientation.

Method 

Twenty-five gay and 26 straight German speakers provided data for a fine-grained psychological self-assessment (e.g., masculinity/femininity) and explicit speech stereotypes. They were recorded for an extensive set of read and spontaneous speech samples using microphones and nasometry. Recordings were analyzed for a variety of acoustic parameters (e.g., fundamental frequency and nasalance). Seventy-four listeners categorized speakers as gay or straight on the basis of the same sentence.

Results 

Most relevant explicitly expressed speech stereotypes encompass voice pitch, nasality, chromaticity, and smoothness. Demonstrating implicit stereotypes, speakers were perceived as sounding straighter, the lower their median f0, center of gravity in /s/, and mean F2. However, based on actual sexual orientation, straight men only showed lower mean F1 than gay men. Additionally, we found evidence that actual masculinity/femininity and the degree of sexual orientation were reflected in gay and straight men's speech.

Conclusion 

Implicit and explicit speech stereotypes about gay and straight men do not contain a kernel of truth, and differences within groups are more important than differences between them.

Supplemental Material 

https://doi.org/10.23641/asha.6484001},
	number = {7},
	urldate = {2024-02-26},
	journal = {Journal of Speech, Language, and Hearing Research},
	publisher = {American Speech-Language-Hearing Association},
	author = {Kachel, Sven and Simpson, Adrian P. and Steffens, Melanie C.},
	month = jul,
	year = {2018},
	pages = {1560--1578},
}

@article{elazar_whats_2024,
	title = {What's {In} {My} {Big} {Data}?},
	url = {http://arxiv.org/abs/2310.20707},
	doi = {10.48550/arXiv.2310.20707},
	abstract = {Large text corpora are the backbone of language models. However, we have a limited understanding of the content of these corpora, including general statistics, quality, social factors, and inclusion of evaluation data (contamination). In this work, we propose What's In My Big Data? (WIMBD), a platform and a set of sixteen analyses that allow us to reveal and compare the contents of large text corpora. WIMBD builds on two basic capabilities -- count and search -- at scale, which allows us to analyze more than 35 terabytes on a standard compute node. We apply WIMBD to ten different corpora used to train popular language models, including C4, The Pile, and RedPajama. Our analysis uncovers several surprising and previously undocumented findings about these corpora, including the high prevalence of duplicate, synthetic, and low-quality content, personally identifiable information, toxic language, and benchmark contamination. For instance, we find that about 50\% of the documents in RedPajama and LAION-2B-en are duplicates. In addition, several datasets used for benchmarking models trained on such corpora are contaminated with respect to important benchmarks, including the Winograd Schema Challenge and parts of GLUE and SuperGLUE. We open-source WIMBD's code and artifacts to provide a standard set of evaluations for new text-based corpora and to encourage more analyses and transparency around them.},
	urldate = {2024-04-11},
	journal = {arXiv},
	author = {Elazar, Yanai and Bhagia, Akshita and Magnusson, Ian and Ravichander, Abhilasha and Schwenk, Dustin and Suhr, Alane and Walsh, Pete and Groeneveld, Dirk and Soldaini, Luca and Singh, Sameer and Hajishirzi, Hanna and Smith, Noah A. and Dodge, Jesse},
	month = mar,
	year = {2024},
	note = {http://arxiv.org/abs/2310.20707 Accessed 21/04/2024 Accepted at ICLR 2024 Spotlight},
	keywords = {Computer Science - Computation and Language, Computer Science - Machine Learning},
}

@inproceedings{shelby_sociotechnical_2023,
	address = {Montréal QC Canada},
	title = {Sociotechnical {Harms} of {Algorithmic} {Systems}: {Scoping} a {Taxonomy} for {Harm} {Reduction}},
	isbn = {979-8-4007-0231-0},
	shorttitle = {Sociotechnical {Harms} of {Algorithmic} {Systems}},
	url = {https://dl.acm.org/doi/10.1145/3600211.3604673},
	doi = {10.1145/3600211.3604673},
	language = {en},
	urldate = {2025-02-07},
	booktitle = {Proceedings of the 2023 {AAAI}/{ACM} {Conference} on {AI}, {Ethics}, and {Society}},
	publisher = {ACM},
	author = {Shelby, Renee and Rismani, Shalaleh and Henne, Kathryn and Moon, AJung and Rostamzadeh, Negar and Nicholas, Paul and Yilla-Akbari, N'Mah and Gallegos, Jess and Smart, Andrew and Garcia, Emilio and Virk, Gurleen},
	month = aug,
	year = {2023},
	pages = {723--741},
}

@inproceedings{sigurgeirsson_just_2024,
	title = {Just {Because} {We} {Camp}, {Doesn}'t {Mean} {We} {Should}: {The} {Ethics} of {Modelling} {Queer} {Voices}.},
	shorttitle = {Just {Because} {We} {Camp}, {Doesn}'t {Mean} {We} {Should}},
	url = {https://www.isca-archive.org/interspeech_2024/sigurgeirsson24_interspeech.html},
	doi = {10.21437/Interspeech.2024-1982},
	abstract = {Modern voice cloning models claim to be able to capture a diverse range of voices. We test the ability of a typical pipeline to capture the style known colloquially as “gay voice” and notice a homogenisation effect: synthesised speech is rated as sounding significantly “less gay” (by LGBTQ+ participants) than its corresponding ground-truth for speakers with “gay voice”, but ratings actually increase for control speakers. Loss of “gay voice” has implications for accessibility. We also find that for speakers with “gay voice”, loss of “gay voice” corresponds to lower similarity ratings.},
	language = {en},
	urldate = {2025-03-26},
	booktitle = {Interspeech 2024},
	publisher = {ISCA},
	author = {Sigurgeirsson, Atli and Ungless, Eddie L.},
	month = sep,
	year = {2024},
	pages = {3050--3054},
	file = {PDF:/Users/eddieungless/Zotero/storage/AHJH6VAW/Sigurgeirsson and Ungless - 2024 - Just Because We Camp, Doesn't Mean We Should The Ethics of Modelling Queer Voices..pdf:application/pdf},
}

@inproceedings{ungless_amplifying_2025,
	address = {Vienna, Austria},
	title = {Amplifying {Trans} and {Nonbinary} {Voices}: {A} {Community}-{Centred} {Harm} {Taxonomy} for {LLMs}},
	isbn = {979-8-89176-251-0},
	shorttitle = {Amplifying {Trans} and {Nonbinary} {Voices}},
	url = {https://aclanthology.org/2025.acl-long.1001/},
	doi = {10.18653/v1/2025.acl-long.1001},
	abstract = {We explore large language model (LLM) responses that may negatively impact the transgender and nonbinary (TGNB) community and introduce the Transing Transformers Toolkit, T{\textasciicircum}3, which provides resources for identifying such harmful response behaviors. The heart of T{\textasciicircum}3 is a community-centred taxonomy of harms, developed in collaboration with the TGNB community, which we complement with, amongst other guidance, suggested heuristics for evaluation. To develop the taxonomy, we adopted a multi-method approach that included surveys and focus groups with community experts. The contribution highlights the importance of community-centred approaches in mitigating harm, and outlines pathways for LLM developers to improve how their models handle TGNB-related topics.},
	urldate = {2025-08-21},
	booktitle = {Proceedings of the 63rd {Annual} {Meeting} of the {Association} for {Computational} {Linguistics} ({Volume} 1: {Long} {Papers})},
	publisher = {Association for Computational Linguistics},
	author = {Ungless, Eddie L. and Dev, Sunipa and Bennett, Cynthia L. and Gulotta, Rebecca and Bastings, Jasmijn and Denton, Remi},
	editor = {Che, Wanxiang and Nabende, Joyce and Shutova, Ekaterina and Pilehvar, Mohammad Taher},
	month = jul,
	year = {2025},
	pages = {20503--20535},
	file = {Full Text PDF:/Users/eddieungless/Zotero/storage/N756DCLA/Ungless et al. - 2025 - Amplifying Trans and Nonbinary Voices A Community-Centred Harm Taxonomy for LLMs.pdf:application/pdf},
}

@inproceedings{buolamwini_gender_2018,
	title = {Gender {Shades}: {Intersectional} {Accuracy} {Disparities} in {Commercial} {Gender} {Classification}},
	shorttitle = {Gender {Shades}},
	url = {https://proceedings.mlr.press/v81/buolamwini18a.html},
	language = {en},
	urldate = {2021-09-23},
	booktitle = {Conference on {Fairness}, {Accountability} and {Transparency}},
	publisher = {PMLR},
	author = {Buolamwini, Joy and Gebru, Timnit},
	month = jan,
	year = {2018},
	pages = {77--91},
	file = {Buolamwini_Gebru_2018_Gender Shades.pdf:/Users/eddieungless/Zotero/storage/3IBDMUSU/Buolamwini_Gebru_2018_Gender Shades.pdf:application/pdf;Buolamwini_Gebru_2018_Gender Shades.pdf:/Users/eddieungless/Zotero/storage/XZQPCQGF/Buolamwini_Gebru_2018_Gender Shades2.pdf:application/pdf;Supplementary PDF:/Users/eddieungless/Zotero/storage/76RLA4BG/Buolamwini and Gebru - 2018 - Gender Shades Intersectional Accuracy Disparities.pdf:application/pdf},
}

@article{hooker_moving_2021,
	title = {Moving beyond “algorithmic bias is a data problem”},
	volume = {2},
	issn = {2666-3899},
	url = {https://www.sciencedirect.com/science/article/pii/S2666389921000611},
	doi = {10.1016/j.patter.2021.100241},
	abstract = {A surprisingly sticky belief is that a machine learning model merely reflects existing algorithmic bias in the dataset and does not itself contribute to harm. Why, despite clear evidence to the contrary, does the myth of the impartial model still hold allure for so many within our research community? Algorithms are not impartial, and some design choices are better than others. Recognizing how model design impacts harm opens up new mitigation techniques that are less burdensome than comprehensive data collection.},
	number = {4},
	urldate = {2024-06-10},
	journal = {Patterns},
	author = {Hooker, Sara},
	month = apr,
	year = {2021},
	pages = {100241},
	file = {Full Text:/Users/eddieungless/Zotero/storage/JSH9QRB9/Hooker - 2021 - Moving beyond “algorithmic bias is a data problem”.pdf:application/pdf;ScienceDirect Snapshot:/Users/eddieungless/Zotero/storage/MI5LY5GZ/S2666389921000611.html:text/html},
}

@inproceedings{ungless_stereotypes_2023,
	address = {Toronto, Canada},
	title = {Stereotypes and {Smut}: {The} ({Mis})representation of {Non}-cisgender {Identities} by {Text}-to-{Image} {Models}},
	shorttitle = {Stereotypes and {Smut}},
	url = {https://aclanthology.org/2023.findings-acl.502/},
	doi = {10.18653/v1/2023.findings-acl.502},
	abstract = {Cutting-edge image generation has been praised for producing high-quality images, suggesting a ubiquitous future in a variety of applications. However, initial studies have pointed to the potential for harm due to predictive bias, reflecting and potentially reinforcing cultural stereotypes. In this work, we are the first to investigate how multimodal models handle diverse gender identities. Concretely, we conduct a thorough analysis in which we compare the output of three image generation models for prompts containing cisgender vs. non-cisgender identity terms. Our findings demonstrate that certain non-cisgender identities are consistently (mis)represented as less human, more stereotyped and more sexualised. We complement our experimental analysis with (a) a survey among non-cisgender individuals and (b) a series of interviews, to establish which harms affected individuals anticipate, and how they would like to be represented. We find respondents are particularly concerned about misrepresentation, and the potential to drive harmful behaviours and beliefs. Simple heuristics to limit offensive content are widely rejected, and instead respondents call for community involvement, curated training data and the ability to customise. These improvements could pave the way for a future where change is led by the affected community, and technology is used to positively “[portray] queerness in ways that we haven't even thought of‴ rather than reproducing stale, offensive stereotypes.},
	urldate = {2026-05-28},
	booktitle = {Findings of the {Association} for {Computational} {Linguistics}: {ACL} 2023},
	publisher = {Association for Computational Linguistics},
	author = {Ungless, Eddie and Ross, Bjorn and Lauscher, Anne},
	editor = {Rogers, Anna and Boyd-Graber, Jordan and Okazaki, Naoaki},
	month = jul,
	year = {2023},
	pages = {7919--7942},
	file = {Full Text PDF:/Users/eddieungless/Zotero/storage/NZP639MI/Ungless et al. - 2023 - Stereotypes and Smut The (Mis)representation of Non-cisgender Identities by Text-to-Image Models.pdf:application/pdf},
}

@inproceedings{chen_surfacing_2026,
	address = {New York, NY, USA},
	series = {{CHI} '26},
	title = {Surfacing and {Applying} {Meaning}: {Supporting} {Hermeneutical} {Autonomy} for {LGBTQ}+ {People} in {Taiwan}},
	isbn = {979-8-4007-2278-3},
	shorttitle = {Surfacing and {Applying} {Meaning}},
	url = {https://dl.acm.org/doi/10.1145/3772318.3790816},
	doi = {10.1145/3772318.3790816},
	abstract = {After Taiwan’s legalization of same-sex marriage in 2019, LGBTQ+ communities continue to face hostility on social media. Using the lens of hermeneutical injustice and autonomy, we examine how technological conditions affect LGBTQ+ individuals’ identity exploration, narrative seeking, and community resilience. We conducted a multi-stage study with Taiwanese LGBTQ+ individuals, including in-depth interviews, participatory design workshops, and evaluation sessions. Participants described fragile yet creative strategies such as seeking validation in online interactions, reframing hostile content through theory, and relying on allies. Building on these insights, we designed and evaluated a retrieval-augmented, LLM-powered chatbot with four modes of interaction: reflection, validation, discussion, and allyship. Findings show that the system fosters hermeneutical autonomy by helping participants reframe hostile narratives, validate lived experiences, and scaffold identity exploration, while reducing the hermeneutical labor of navigating social media hostility. We conclude by outlining design implications for AI systems that advance hermeneutical autonomy through fluid self-representation, contextualized dialogue, and inclusive community participation.},
	urldate = {2026-05-28},
	booktitle = {Proceedings of the 2026 {CHI} {Conference} on {Human} {Factors} in {Computing} {Systems}},
	publisher = {Association for Computing Machinery},
	author = {Chen, Yi-Tong and Chang, En-Kai and Bi, Nanyi and Goyal, Nitesh},
	month = apr,
	year = {2026},
	pages = {1--28},
	file = {Full Text PDF:/Users/eddieungless/Zotero/storage/DU7KZP74/Chen et al. - 2026 - Surfacing and Applying Meaning Supporting Hermeneutical Autonomy for LGBTQ+ People in Taiwan.pdf:application/pdf},
}
