@inproceedings{bibcite_16157, author = {Anna Zawadzka and Przemys{\l}aw G{\l}omb}, title = {Architecture Matters: Gender Disparities in Automated Image Moderation}, abstract = {
Automated image moderation systems shape online visibility and dataset curation, yet prior work has identified demographic disparities in similarity-based NSFW classifiers. We compare such systems with instruction-tuned vision-language models (VLMs) that generate structured moderation decisions with textual rationales. Using the PHASE-annotated subset of the GCC dataset, \ we compute false positive rates overall and by gender.\
Results show substantial variation in moderation strictness and in bias direction: two models show more pronounced strictness in moderation of male images, while other two exhibit higher removal rates for female images.\
Reasoning-based models do not consistently mitigate bias but, in the proposed pipeline, they offer a transparency layer.