<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.0 20120330//EN" "JATS-archivearticle1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta>
      <journal-title-group>
        <journal-title>Forum for Information Retrieval Evaluation, December</journal-title>
      </journal-title-group>
    </journal-meta>
    <article-meta>
      <title-group>
        <article-title>of the HASOC Track at FIRE 2025: Abusive Meme Identification - Shadows Behind the Laughter</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <string-name>Koyel Ghosh</string-name>
          <email>koyelg@srmist.edu.in</email>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Mithun Das</string-name>
          <email>mithun.rcciit@gmail.com</email>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Shubhankar Barman</string-name>
          <email>contact.shubhankarbarman@gmail.com</email>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Mwnthai Narzary</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Saptarshi Saha</string-name>
          <email>Saptarshi2016saha@gmail.com</email>
          <xref ref-type="aff" rid="aff3">3</xref>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Animesh Mukherjee</string-name>
          <email>animeshm@gmail.com</email>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Sandip Modha</string-name>
          <email>sjmodha@gmail.com</email>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Debasis Ganguly</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Utpal Garain</string-name>
          <email>utpal.garain@gmail.com</email>
          <xref ref-type="aff" rid="aff3">3</xref>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Sylvia Jaki</string-name>
          <email>sylvia.jaki@kuleuven.be</email>
          <xref ref-type="aff" rid="aff4">4</xref>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Thomas Mandl</string-name>
          <email>mandl@uni-hildesheim.de</email>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>BITS Pilani</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>India</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="editor">
          <string-name>Hate Speech, Abusive Meme Identification, Social NLP, Social Media, Memes, Deep Learning, Low-Resource</string-name>
        </contrib>
        <aff id="aff0">
          <label>0</label>
          <institution>Central Institute of Technology</institution>
          ,
          <addr-line>Kokrajhar</addr-line>
          ,
          <country country="IN">India</country>
        </aff>
        <aff id="aff1">
          <label>1</label>
          <institution>Dhirubhai Ambani University</institution>
          ,
          <addr-line>Gandhinagar</addr-line>
          ,
          <country country="IN">India</country>
        </aff>
        <aff id="aff2">
          <label>2</label>
          <institution>Indian Institute of Technology</institution>
          ,
          <addr-line>Kharagpur</addr-line>
          ,
          <country country="IN">India</country>
        </aff>
        <aff id="aff3">
          <label>3</label>
          <institution>Indian Statistical Institute</institution>
          ,
          <addr-line>Kolkata</addr-line>
          ,
          <country country="IN">India</country>
        </aff>
        <aff id="aff4">
          <label>4</label>
          <institution>Katholieke Universiteit Leuven</institution>
          ,
          <addr-line>Campus Antwerp</addr-line>
          ,
          <country country="BE">Belgium</country>
        </aff>
        <aff id="aff5">
          <label>5</label>
          <institution>Language</institution>
          ,
          <addr-line>Indian Language, Benchmark, Bangla, Hindi, Gujarati, Bodo, HASOC</addr-line>
        </aff>
        <aff id="aff6">
          <label>6</label>
          <institution>SRM Institute of Science &amp; Technology</institution>
          ,
          <addr-line>Kattankulathur</addr-line>
          ,
          <country country="IN">India</country>
        </aff>
        <aff id="aff7">
          <label>7</label>
          <institution>University of Glasgow</institution>
          ,
          <country country="UK">United Kingdom</country>
        </aff>
        <aff id="aff8">
          <label>8</label>
          <institution>University of Hildesheim</institution>
          ,
          <addr-line>Hildesheim</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff9">
          <label>9</label>
          <institution>University of Milano-Bicocca</institution>
          ,
          <addr-line>Milan</addr-line>
          ,
          <country country="IT">Italy</country>
        </aff>
      </contrib-group>
      <pub-date>
        <year>2025</year>
      </pub-date>
      <volume>1</volume>
      <fpage>7</fpage>
      <lpage>20</lpage>
      <abstract>
        <p>The dramatic surge in the use of social media platforms for sharing information and opinions has, unfortunately, also fueled a sharp rise in online abuse. One of the simplest yet most insidious tools for such abuse is the meme: a visual artifact that typically fuses an image with a short, often provocative, text overlay. While memes can be humorous or satirical, they are increasingly being weaponized to target individuals or communities through hateful or derogatory content. This poses a serious threat to online safety and digital well-being. Detecting and curbing such abusive memes is therefore an urgent necessity. However, the task becomes significantly more challenging in low-resource Indian languages such as Bangla, Hindi, Gujarati, Bodo etc., where annotated benchmark datasets are scarce or entirely absent. Without such datasets, the development and evaluation of robust AI models remain severely constrained. In this work, we aim to bridge this crucial gap by constructing a multilingual abusive meme dataset-covering Bangla, Hindi, Gujarati, and Bodo, thereby enabling research and innovation in abusive meme detection across low-resource Indian languages. Each dataset is annotated with five classification labels: sentiment, sarcasm, vulgarity, abuse, and target. A total of 20 unique teams participated, submitting over 306 runs across the four languages. The performance of the best systems was evaluated using the Macro F1 score, with the top-performing models achieving scores of 0.6275 (Bangla), 0.6570 (Hindi), 0.6750 (Gujarati), and 0.6312 (Bodo). This article briefly summarizes the tasks, data development, results and approaches.</p>
      </abstract>
      <kwd-group>
        <kwd>Laughter</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec id="sec-1">
      <title>1. Introduction</title>
      <p>
        The concept of memes was first proposed by Richard Dawkins in 1976, describing them as cultural
elements that spread from person to person much like genes, quickly sharing ideas and influencing
group thinking [
        <xref ref-type="bibr" rid="ref1">1</xref>
        ]. Yet, memes can also pose certain risks that deeply afect both people and society
      </p>
      <p>CEUR
Workshop</p>
      <p>
        ISSN1613-0073
as a whole. In today’s world, “Internet memes” or “image memes” are widely seen on social media
platforms. A meme is typically a visual idea that combines an image with a short caption written over
it, making the text an essential part of the picture [
        <xref ref-type="bibr" rid="ref2">2</xref>
        ]. Internet memes have become powerful means of
communication and self-expression, quickly spreading ideas, feelings, and cultural meanings across
online communities. While they are often created for fun and humor, memes also play an important
role in shaping public opinion and political discussions. Although usually intended as jokes, in today’s
online world, memes go far beyond simple humor. People create them easily and at no cost, using them
to attract attention and gain social recognition. The growing presence of memes on social media has
raised both curiosity and concern about their influence on society, as they play a major part in digital
culture and mirror the shared mindset of online communities. Malicious users frequently employ memes
to intimidate or harm individuals or specific groups. This form of content is termed abusive memes.
Owing to their rapid shareability, such memes can escalate social conflicts, harm the credibility of online
platforms [
        <xref ref-type="bibr" rid="ref3">3</xref>
        ], and cause significant emotional distress to afected individuals [
        <xref ref-type="bibr" rid="ref4">4</xref>
        ]. Hence, it is essential
to regulate their dissemination, and the primary step in this direction is the accurate identification of
abusive memes. Memes have been found to be highly efective in spreading hateful ideologies, due
to their seemingly humorous undertone. Research has shown that hateful memes can be surprisingly
persuasive, often going viral and resonating with a wide audience [
        <xref ref-type="bibr" rid="ref5">5</xref>
        ]. This phenomenon highlights
the need to examine the role of memes in perpetuating hateful ideologies and their potential impact
on society. Memes can develop a highly persuasive efect due to their seemingly humorous undertone
[
        <xref ref-type="bibr" rid="ref6">6</xref>
        ]. In recent years, several studies have attempted to detect and mitigate the impact of abusive memes
on social media. However, most of this research has focused primarily on memes containing English
text [
        <xref ref-type="bibr" rid="ref10 ref11 ref7 ref8 ref9">7, 8, 9, 10, 11</xref>
        ]. Moreover, multiple multimodal vision language models have been investigated, but
most of them remain confined to English. Research eforts in other languages are still limited, especially
for low-resource Indian languages such as Bangla, Hindi, Gujarati, Bodo, etc.
      </p>
      <p>
        The HASOC (Hate Speech and Ofensive Content Identification) track, which has provided a platform
for hate speech detection since 2019 at FIRE (Forum for Information Retrieval Evaluation) [
        <xref ref-type="bibr" rid="ref12">12</xref>
        ]. This
year, HASOC 2025 provides a comprehensive platform for advancing research on hate speech and
ofensive content across multiple modalities and languages. In this context, the present paper ofers
an overview of abusive meme identification datasets available in Bangla, Hindi, Gujarati, and Bodo.
These datasets are crucial, as they contribute valuable task-specific resources for languages that are
traditionally low-resourced. By compiling and analyzing these meme datasets, the paper aims to
support the development of more robust and inclusive multimodal hate speech detection systems. Here,
we present the task, dataset, annotation process, participants’ results, as well as their short system
descriptions.
      </p>
    </sec>
    <sec id="sec-2">
      <title>2. Related Work</title>
      <sec id="sec-2-1">
        <title>2.1. Multimodal Abusive Meme Datasets</title>
        <p>
          Research on multimodal abusive or hateful memes has grown steadily, supported by the development
of several datasets over the last few years. One of the earliest initiatives was by Sabat et al. [
          <xref ref-type="bibr" rid="ref13">13</xref>
          ], who
collected 5,020 memes from Google Images to study hate meme detection. This was followed by the
large scale MMHS150K dataset by Gomez et al. [
          <xref ref-type="bibr" rid="ref14">14</xref>
          ], consisting of 150K meme like posts gathered from
Twitter. Chandra et al. [
          <xref ref-type="bibr" rid="ref15">15</xref>
          ] broadened the scope by constructing a dataset dedicated to antisemitic
memes using Twitter and Gab. Likewise, Suryawanshi et al. [
          <xref ref-type="bibr" rid="ref16">16</xref>
          ] compiled 743 election related memes
annotated for ofensiveness, while Pramanick et al. [
          <xref ref-type="bibr" rid="ref17">17</xref>
          ] introduced a dataset of around 3.5K harmful
COVID-19 memes.
        </p>
        <p>
          A major milestone in this space is the Hateful Memes Challenge dataset created by Facebook AI [
          <xref ref-type="bibr" rid="ref18">18</xref>
          ].
It marked the beginning of large-scale, carefully curated multimodal hate speech datasets. The authors
emphasized the context-dependent nature of memes: an identical image paired with diferent text
may or may not be hateful. To capture this phenomenon, the dataset includes manually constructed
confounders, in which either the image or the text was replaced with a benign counterpart. This
prevents models from exploiting superficial correlations, although it also results in memes that may not
naturally occur on social media.
        </p>
        <p>
          The release of this dataset catalyzed a series of multimodal shared tasks. SemEval-2022 Task 5
(MAMI) [
          <xref ref-type="bibr" rid="ref19">19</xref>
          ] focused on misogynistic multimodal content, while EXIST 2024 [
          <xref ref-type="bibr" rid="ref20">20</xref>
          ]1 explored abusive and
aggressive communication across languages. Pandiani et al. [
          <xref ref-type="bibr" rid="ref21">21</xref>
          ] ofer an overview of 34 multimodal
toxic content datasets, highlighting that most resources remain English centric and that many world
languages lack suficient multimodal hate-speech data.
        </p>
      </sec>
      <sec id="sec-2-2">
        <title>2.2. Datasets for Indian Languages</title>
        <p>
          In contrast to English, multimodal meme datasets for Indian languages are limited and often not publicly
available. Karim et al. [
          <xref ref-type="bibr" rid="ref22">22</xref>
          ] expanded the Bengali hate speech dataset [
          <xref ref-type="bibr" rid="ref22">22</xref>
          ] by labelling 4,500 memes,
but the dataset is not publicly released, and details about data collection, annotation protocols, target
communities, and inter-annotator agreement are lacking. Hossain et al. [
          <xref ref-type="bibr" rid="ref23">23</xref>
          ] developed another Bengali
dataset containing 4,158 memes with Bengali and code-mixed captions. However, this dataset too
remains unavailable, and critical labels, such as target groups are missing. Das et al. [
          <xref ref-type="bibr" rid="ref24">24</xref>
          ], their dataset
contains 4,043 Bengali memes. Among them, 1,515 are tagged as abusive and 2,528 as non-abusive. A
total of 1,664 memes are marked as sarcastic, while 2,379 are not. For vulgarity, 1,171 memes are labeled
vulgar and 2,872 non-vulgar. In terms of sentiment, 592 express positive sentiment, 1,414 are neutral,
and 2,037 convey negative sentiment. These limitations echo the concerns raised by Kirk et al. [
          <xref ref-type="bibr" rid="ref25">25</xref>
          ],
who note that memes in research datasets often fail to represent the diversity, stylistic variation, and
informality found in real social media memes.
        </p>
        <p>
          Emofmeme [
          <xref ref-type="bibr" rid="ref26">26</xref>
          ] is a Hindi meme dataset comprising 7,500 samples designed for ofensive meme
detection, where ofensiveness is annotated using a binary (yes/no) labeling scheme. In addition,
the dataset captures the emotional or sentiment aspect of memes through a multi-label, multi-class
framework, covering emotions such as fear, neglect, irritation, rage, disgust, nervousness, shame,
disappointment, envy, sufering, sadness, joy, pride, and surprise. The Indian Political Memes (IPM)
dataset [
          <xref ref-type="bibr" rid="ref27">27</xref>
          ] contains 1,218 Hindi and English memes focused on hateful meme detection. Unlike binary
schemes, IPM adopts a single-label multi-class annotation distinguishing between non-ofensive,
hateinducing, and satirical memes. MultiBully [
          <xref ref-type="bibr" rid="ref28">28</xref>
          ] includes 5,854 Hindi and English memes for cyberbullying
detection, while MultiBully-Ex [
          <xref ref-type="bibr" rid="ref29">29</xref>
          ] extends it with 3,222 memes focused on cyberbullying explanation.
Both datasets annotate bullying using binary labels (bully/non-bully) and further apply single-label
multi-class schemes for emotions (e.g., joy, sadness, fear, anger, disgust, surprise, anticipation, trust, and
ridicule) as well as coarse sentiment categories (positive, neutral, negative). In addition, they classify
the degree of harmfulness into very harmful, partially harmful, and harmless using a single-label
multi-class setup, enabling a finer-grained analysis of harmful content. Irony and sarcasm are annotated
in binary form in the MultiBully [
          <xref ref-type="bibr" rid="ref28">28</xref>
          ] and MultiBully-Ex [
          <xref ref-type="bibr" rid="ref29">29</xref>
          ], respectively. Pol_Of_Meme [ 30] is a
Hindi and English dataset of 7,500 memes for ofensive meme detection. It uses binary labels (yes/no) to
mark ofensiveness, further distinguishes implicit and explicit ofensiveness through binary annotation,
and captures emotions via a multi-label multi-class scheme covering fear, neglect, irritation, rage,
disgust, nervousness, shame, disappointment, envy, sufering, sadness, joy, pride, and surprise; political
attributes are also identified using a binary label indicating whether a meme is political or not.
        </p>
        <p>TamilMemes [31] is a Tamil-language dataset containing 2,969 memes created for troll meme detection.
In this dataset, trolling is annotated as a dedicated dimension using a binary labeling scheme (yes/no).</p>
        <p>While most existing eforts focus on English, abusive meme detection in Indian languages remains
largely underexplored, with the exception of Bengali and Hindi. To bridge this gap, our work presents a
multimodal abusive-meme dataset for Bangla, Hindi, Gujarati, and Bodo, designed to closely reflect
memes typically shared on real social media platforms. The scarcity of datasets, limited transparency,
and lack of resources that capture real-world meme distributions strongly motivate the development of
high-quality datasets for Indian languages, which forms the core motivation of this study.</p>
      </sec>
    </sec>
    <sec id="sec-3">
      <title>3. Task Description</title>
      <p>In HASOC meme 2025, the task is with four languages proposed in the research area of hate speech
detection. These tasks ofered all four languages: Bangla, Hindi, Gujarati, and Bodo. Figure 1 shows the
Screenshot of HASOC meme Website 2.</p>
      <p>This task involves analyzing multimodal data (image and text) to detect abuse, identify targeted
communities, assess vulgarity and sarcasm, and assign sentiment labels. So, the task will be in five
parts. (1) Sentiment detection: positive, negative, and neutral, (2) Sarcasm Detection: sarcastic or not,
(3) Vulgarity Detection: vulgar or not, (4) Abuse Detection: abusive or not, and (5) Target Community
Identification: Gender, Religion, Individual, Political, National Origin, Social Sub-groups, Others, and
none.</p>
    </sec>
    <sec id="sec-4">
      <title>4. Dataset Description</title>
      <p>In this section, dataset collection, annotation, and analysis have been discussed for HASOC-meme.</p>
      <sec id="sec-4-1">
        <title>4.1. Dataset Collection</title>
        <p>
          Our primary aim in constructing this multilingual, multimodal dataset is to ensure its diversity, so we
intentionally selected some Gender, Religion, Political, communal ™Facebook meme pages, ™Instagram,
™Google Image, ™Bing, and ™YouTube channels etc. Unlike the Hateful Memes Challenge [
          <xref ref-type="bibr" rid="ref18">18</xref>
          ],
which used synthetically generated memes, our dataset is built entirely from real world memes. After
collecting the images, we performed several filtering steps before annotation. Memes without text,
memes containing text in languages other than the targeted four languages, and memes with very
low resolution that made the text unreadable were removed. These same preprocessing steps were
consistently applied across all four languages: Bangla, Hindi, Gujarati, and Bodo to ensure clean and
reliable datasets.
        </p>
      </sec>
      <sec id="sec-4-2">
        <title>4.2. Dataset Annotation</title>
        <p>For the annotation process, we engaged fourteen undergraduate students (aged 20–25 years), consisting
of nine males and five females, to identify whether each meme was abusive or non-abusive. These
2https://hasocfire.github.io/hasoc/2025/call_for_participation.html (Access on 05.12.2025)</p>
        <sec id="sec-4-2-1">
          <title>Sarcasm</title>
        </sec>
        <sec id="sec-4-2-2">
          <title>Vulgarity</title>
        </sec>
        <sec id="sec-4-2-3">
          <title>Abuse</title>
        </sec>
        <sec id="sec-4-2-4">
          <title>Target Community</title>
          <p>Determine whether
the meme conveys
sarcasm or irony
Identify vulgar or
obscene
language/imagery
Determine whether
the meme is abusive</p>
        </sec>
        <sec id="sec-4-2-5">
          <title>Objective Labels / Description</title>
          <p>Assign a sentiment la- Positive: Supportive, humorous, or appreciative tone
bel to the meme Neutral: Neither positive nor negative</p>
          <p>Negative: Hostility, mockery, or criticism
Sarcastic: The meme conveys sarcasm or irony
Non-Sarcastic: The meme does not convey sarcasm or
irony
Vulgar: Explicit or ofensive words/gestures
Not Vulgar: No explicit or ofensive content
Abusive: Harmful, ofensive, or derogatory content
targeting individuals or groups</p>
          <p>Non-Abusive: No harmful or derogatory content
Identify the commu- Gender: Male, female, non-binary, transgender
nity targeted by the Religion: Religious beliefs, deities, practices
meme, if any Individual: Targeting a specific person.</p>
          <p>Political: Political ideologies, parties, people
National Origin: Country or ethnicity-based groups
Social Sub-groups: Socio-economic, cultural,
occupational groups
Others: Any target not in above categories</p>
          <p>
            None: No specific target
annotators worked across all four languages, like Bangla, Hindi, Gujarati, and Bodo, and were supported
with OCR-extracted text. All students were native speakers of the respective languages and were
compensated fairly according to local standards. The entire annotation workflow was overseen by a
post-doctoral researcher, Ph.D. researchers, and an industry-academic researcher, each with several
years of experience in analyzing harmful social media content. We adopt the same schemes as the
authors [
            <xref ref-type="bibr" rid="ref24">24</xref>
            ] for the annotation rules with little necessary updates. Our framework includes six types of
annotations, as presented in the table 1 below.
          </p>
        </sec>
      </sec>
      <sec id="sec-4-3">
        <title>4.3. Dataset Analysis</title>
        <p>Figure 2 shows the train and test split of the Bangla, Hindi, Gujarati, and Bodo datasets. We summarize
the key statistics of our multilingual meme traning datasets in Table 2. Across all four languages,
noticeable label imbalance is present in every annotation category. In the Bangla dataset, negative
sentiment (1,476 samples) is substantially higher than positive (906) and neutral (311) classes. Hindi
exhibits a similar pattern, where negative sentiment dominates the other two classes. In the Gujarati,
more positive and neutral labels are present than negative. The Bodo dataset is particularly imbalanced,
containing no neutral samples and showing a clear skew toward the positive class (227 positive vs. 151
negative).</p>
        <p>A strong imbalance is also observed in sarcasm labels. Bangla contains a significantly higher
proportion of sarcastic memes (2,081) compared to non-sarcastic ones (612). Hindi (770 vs. 371) and Gujarati
(670 vs. 219) show similar skewed distributions. In the Bodo dataset, this imbalance becomes even more
pronounced, with 339 sarcastic and only 39 non-sarcastic samples.</p>
        <p>Vulgarity labels show imbalance in the opposite direction for most languages. All four datasets have
considerably more non-vulgar than vulgar samples. For instance, Bangla (2,226 non-vulgar vs. 467
vulgar), Hindi (764 vs. 377), Gujarati (592 vs. 297), and Bodo (271 vs. 107), indicating that vulgar memes
represent a minority class.</p>
        <p>(a)
(b)</p>
        <p>Abuse labels also display asymmetry. While Bangla and Hindi contain more non-abusive than abusive
samples, the gap is large (Bangla: 1,954 vs. 739; Hindi: 834 vs. 307). Gujarati follows the same trend,
whereas Bodo has a relatively lower but still imbalanced distribution (381 non-abusive vs. 77 abusive).</p>
        <p>Overall, each of the four languages demonstrates clear class imbalance across sentiment, sarcasm,
vulgarity, and abuse labels, which highlights the need for careful modeling strategies and balanced
evaluation. Figure 3 provides a graphical visualization of these distributions.</p>
      </sec>
    </sec>
    <sec id="sec-5">
      <title>5. Results</title>
      <p>The macro F1-score computes the F1-score independently for each label: sentiment, abuse, vulgarity,
sarcasm, then averages them. This metric places greater emphasis on minority classes, penalizing
systems more heavily when performance is weak on underrepresented labels. The choice of an F1
variant depends on the task objectives and class distribution; however, due to the significant class
imbalance typically present in hate speech detection, the macro F1-score is the most appropriate
evaluation measure. For participant’s system run submissions and evaluation in our task, we rely on
the Kaggle platform. Figure 4 displays a screenshot of the Kaggle leaderboard used for run submissions.
Separate Kaggle competition pages are provided for Bangla 3, Hindi 4, Gujarati 5 and Bodo 6 to enable
participants to submit their experimental results.</p>
      <p>Overall, 45 participants register for the task HASOC meme. In the Bangla task, 17 teams made
69 submissions, while 18 teams submitted 82 runs in the Hindi task, for Gujarati 15 teams made 68
submissions, and for the Bodo task, 15 teams submitted a total of 87 runs.</p>
      <p>The performance of the best classification algorithms for Bangla, Hindi, Gujarati, and Bodo are
Macro F1 measures of 0.6275, 0.6570, 0.6750, and 0.6312, respectively. The results for Bangla, Hindi, and
Gujarati datasets are shown in Table 3, Table 4, Table 5, and Table 6, respectively.
3https://www.kaggle.com/competitions/hasoc-2025-meme-bangla (Access on 05.12.2025)
4https://www.kaggle.com/competitions/hasoc-2025-meme-hindi (Access on 05.12.2025)
5https://www.kaggle.com/competitions/hasoc-2025-meme-gujarati (Access on 05.12.2025)
6https://www.kaggle.com/competitions/hasoc-2025-meme-bodo (Access on 05.12.2025)
2500
2000
1500
1000
500</p>
      <p>0
2000
1500
1000
500
0
Positive</p>
      <p>Negative</p>
      <p>Neutral</p>
    </sec>
    <sec id="sec-6">
      <title>6. Methodology</title>
      <p>This section discusses the systems utilized by the participants.</p>
      <p>• FiRC-NLP [32] explored three distinct approaches: prompt-based inference (Zero-Shot
Classification and Few-Shot In-Context Learning via retrieval) using Gemini (Google’s Gemini 2.5 Flash),
cross-modality encoders that combine image and text, and a text-only modality leveraging Optical
Character Recognition (OCR), OCR-based English translation, and image descriptions. Their final
ensemble system fuses these modalities, demonstrating the efectiveness of multimodal fusion
and robust training strategies.
• CSIS BITS Pilani [33] evaluated multiple dual-encoder architectures, combining the CLIP Vision
Transformer with language-specific text models including MuRIL, XLM-Roberta, and M-BERT.
Training is conducted using a 5-fold cross-validation strategy with a weighted loss function to
counteract class imbalance, and performance is validated on the test set.
• Golden Ratio [34] integrated the CLIP vision encoder with tailored Transformer-based language
models: MuRIL for Bangla, XLM-RoBERTa for Hindi, and Bodo. Textual and visual features are
fused using a cross-attention mechanism to enhance classification accuracy.
• SCaLAR [35] experimented with two architectures: a CLIP-based model combining CLIP ViT
visual features with multilingual Sentence Transformer embeddings, and a ViT + XLM-Roberta
model where visual and textual features were concatenated for classification. The CLIP-based
model achieved superior performance in Bangla and Gujarati, underscoring the efectiveness of
contrastive pretraining for aligning visual and linguistic representations.
• NLPFusion [36] For Bangla memes, authors employed ConvNeXt-small, a state-of-the-art
convolutional architecture inspired by transformers, to extract robust semantic visual features. ResNet-34
is used for Hindi and Bodo memes, while ResNet-18 is applied for Gujarati memes. Textual data is
processed using transformer-based models with domain-specific pre-training to handle linguistic
Rank
1
2
3
4
5
6
7
8
9
10
11
12
13
14</p>
      <p>Team
FiRC-NLP [32]
NLPFusion [36]</p>
      <p>MUCS [40]</p>
      <p>SCaLAR [35]
KK_NLP_AI_IIIT_Ranchi [37]</p>
      <p>CSIS BITS Pilani [33]</p>
      <p>IReL [42]
IIT Dhanbad [38]</p>
      <p>CSE_SVNIT [39]
YNU (Kongqiang Wang) [41]</p>
      <p>VEL (Charmathi Rajkumar)
HASOC2025_meme (Baseline)</p>
      <p>DeepSemantics [44]
HASOC_2025 [43]</p>
      <p>diversity and code-mixed content.
• KK_NLP_AI_IIIT_Ranchi [37] first experimented with text-only, low-resource models that
relied solely on OCR-extracted text. They fine-tuned a multilingual IndicBERT backbone and
observed strong performance for Hindi, Gujarati, and Bodo but weak results for Bangla. To
investigate whether the poor Bangla performance was due to IndicBERT limitations, they then
tested monolingual BERT-based models for each language. These included bengali-bert,
hindibert-v2, gujarati-bert, and pretrained-bodo-legal-bert. However, the monolingual backbones did
not yield substantial improvements, and in some cases performed worse than the multilingual
model. Secondly, They incorporated the CLIP Vision model to combine image features with
BERT-based text features using simple feature concatenation. They tested both multilingual
and monolingual text backbones with CLIP, finding improvements for Bangla and Gujarati but
little to no gain for Hindi and consistently poor results for Bodo. Because CLIP failed to extract
meaningful cross-modal features for Indian languages, cross-attention fusion was discarded and
only concatenation-based fusion was used. They then evaluated monolingual BERT backbones
with CLIP, which further improved performance for Bangla, Gujarati, and Hindi. Finally, due to
consistently weak results for Bodo with image features, they submitted a text-only IndicBERT-v2
model for that language.
• IIT Dhanbad [38] developed a multimodal deep learning framework for five subtasks—Sentiment,
Sarcasm, Vulgarity, Abuse, and Target Communities—to automatically classify hateful and
offensive memes. A task-specific strategy was used, where each label was handled by a dedicated
multimodal model using various text encoders (XLM-R, mBERT, MuRIL, BanglaBERT) and image
encoders (EficientNet, DenseNet, VGG19, ResNet), along with techniques like gated attention
and weighted losses. Extensive experiments showed that these task-specific pipelines performed
best when combined through ensemble methods.
• CSE_SVNIT [39] leveraged pre-trained transformer models including XLM-ROBERTa, IndicBERT,
MuRIL, and mBERT enhanced with convolutional neural network (CNN) layers in diferent Indian
languages. Experimental results show that MuRIL provides optimal performance and outperforms
other transformer models in low-resource Indian languages.
• MUCS [40] integrated transformer-based text encoders (Indic-BERT, MuRIL, XLM-Roberta) with
convolutional and transformer-based vision models (ResNet, EficientNet, ViT) using two fusion
mechanisms -concatenation and attention-based strategies, to efectively capture the
complementary cues from both modalities.
• HASOC2025_meme (Baseline) system employs zero-shot prompting using the Qwen2-7B-Instruct
model 7. Without any task-specific fine-tuning, the model is queried directly with carefully
designed prompts to classify memes across the target categories. This baseline demonstrates the
efectiveness of lightweight, instruction-tuned language models in handling multimodal meme
understanding tasks, especially when annotated training data is limited.
• YNU [41] experimented with Gaussian Naive Bayes, Logistic Regression, K-Neighbors Classifier,
Support Vector Machine, Decision Tree Classifier, Linear SVC, Random Forest Classifier, later use
Ensemble Learning.
• IReL [42] developed and evaluated two independent runs under the team name IReL. Run 1
employed XLM-RoBERTa fine-tuned separately for each language (Bangla, Hindi, Gujarati, and
Bodo), leveraging multilingual transformer embeddings to capture semantic nuances in noisy
meme texts. Run 2 applied a zero-shot approach using ChatGPT specifically for Bodo, addressing
the severe lack of annotated resources for this language. Both approaches processed text-based
meme content exclusively.
7https://huggingface.co/Qwen/Qwen2-7B-Instruct (Access on 05.12.2025)
• HASOC_2025 [43] extracted image features using ResNet-101 and combined them with
OCRbased text features from the HASOC-2025 dataset to classify content as abusive/vulgar or not.
They addressed class imbalance using ADASYN and trained separate XGBoost classifiers for
each language. Their experiments showed that this multimodal approach works well across
diferent scripts, even with OCR noise, demonstrating the efectiveness of combining deep image
embeddings with traditional machine learning for multilingual hate speech detection.
• DeepSemantics [44] introduced a multimodal comparative analysis for detecting content that is,
sarcastic, abusive, and vulgar. In addition to that, the sentiment of any content are also captured.
The models are applied on the four diferent datasets consisting of diferent languages: Hindi,
Bodo, Gujarati, and Bengali. Experiments are performed on various models, of which VisualBert
was found to give the best performance. This model is most efective for detecting hateful and
ofensive content in multilingual datasets.
• FAST [46] used (i) a tailored preprocessing pipeline for noisy OCR and Hindi, English code-mixing
using curated stopword and vulgar dictionaries, (ii) a combination of lightweight classical models
(TF–IDF + Random Forest) with neural approaches (CNN, BiLSTM, ResNet50).
• CNLP-UPES [45] proposed a multimodal framework that integrates BERT-based OCR text
embeddings and ResNet-derived image features through an early-fusion strategy, enabling multi-task
predictions across five classification subtasks. To address class imbalance, we incorporate
RandomOverSampler, thereby enhancing the representational balance of minority categories.</p>
    </sec>
    <sec id="sec-7">
      <title>7. Conclusion and Future Work</title>
      <p>In this work, we presented the results of the HASOC Meme 2025 shared task on abusive meme
identification across four languages: Bangla, Hindi, Gujarati, and Bodo. The task attracted a significant
number of participants, reflecting the growing research interest in multimodal hate and abuse detection
for Indian languages. Teams explored a wide range of approaches, including Zero-Shot Classification,
Few-Shot In-Context Learning with retrieval-based prompting, and CLIP-based Vision Transformers,
demonstrating the rapid evolution of multilingual and multimodal learning techniques.</p>
      <p>While the shared task yielded promising insights, it also highlighted substantial gaps, especially for
low-resource languages. Further experimentation is essential, particularly in building richer multimodal
datasets, improving cross-lingual transfer, and developing robust models capable of handling linguistic
diversity and culturally grounded abusive content. Continued eforts in these directions will enable
more inclusive and efective solutions for combating harmful content across a broader spectrum of
languages.</p>
    </sec>
    <sec id="sec-8">
      <title>8. Acknowledgments</title>
      <p>We thank Mr. Atri Chandra, Mr. Farhan Chowdhury, Mr. Animesh Howladar, Mr. Suvankar Dey,
Mr. Trideep Ghosh, Mr. Alankrita Kumari, Mr. Ayush Lodh, Mr. Priyanka Das, Ms. Aveepsa Hatua,
Mr. Bhaskar Pal, Mr. Gajjar Kandarp Dipakkumar, Ms. Vanpariya Palak Govindbhai, Mr. Patel Vraj
Bipinbhai, Ms. Nijira Mushahary for the dataset annotation. We also thank the FIRE and HASOC
organizers for their support in organizing the track. We thank all participants for their submissions and
their valuable work.</p>
    </sec>
    <sec id="sec-9">
      <title>Declaration on Generative AI</title>
      <p>During the preparation of this work, the author(s) used Grammarly in order to: Grammar and spelling
check. After using these tool(s)/service(s), the author(s) reviewed and edited the content as needed and
take(s) full responsibility for the publication’s content.
[30] G. Kumari, A. Sinha, A. Ekbal, A. Chatterjee, V. B. N, Enhancing the fairness of ofensive memes
detection models by mitigating unintended political bias, J. Intell. Inf. Syst. 62 (2024) 735–763.</p>
      <p>URL: https://doi.org/10.1007/s10844-023-00834-9. doi:10.1007/s10844- 023- 00834- 9.
[31] S. Suryawanshi, B. R. Chakravarthi, P. Verma, M. Arcan, J. P. McCrae, P. Buitelaar, A dataset
for troll classification of TamilMemes, in: G. N. Jha, K. Bali, S. L., S. S. Agrawal, A. K. Ojha
(Eds.), Proceedings of the WILDRE5– 5th Workshop on Indian Language Data: Resources and
Evaluation, European Language Resources Association (ELRA), Marseille, France, 2020, pp. 7–13.</p>
      <p>URL: https://aclanthology.org/2020.wildre-1.2/.
[32] F. Hassan, M. S. Jahan, E. Migaev, F. Mtumbuka, Bridging modalities for hate speech detection
in memes, in: Working Notes of FIRE 2025 - Forum for Information Retrieval Evaluation, CEUR,
2025.
[33] R. Bohra, Y. Sharma, A multi-modal ensemble approach for hate speech and ofensive content
detection in indic memes, in: Working Notes of FIRE 2025 - Forum for Information Retrieval
Evaluation, CEUR, 2025.
[34] T. Paul, A. Jamatia, Hate speech and ofensive content identification in memes in bangla, hindi,
and bodo, in: Working Notes of FIRE 2025 - Forum for Information Retrieval Evaluation, CEUR,
2025.
[35] S. S. Nandam, A. K. Madasamy, Multimodal hate speech and ofensive content classification in
code-mixed indian language memes, in: Working Notes of FIRE 2025 - Forum for Information
Retrieval Evaluation, CEUR, 2025.
[36] S. Coelho, A. Hegde, A. M Shetty, Nitte, Hasoc-meme: Enhancing hate speech recognition in
bengali, hindi, gujarati, and bodo memes using multimodal multitask transformers, in: Working
Notes of FIRE 2025 - Forum for Information Retrieval Evaluation, CEUR, 2025.
[37] U. Kedia, S. Prakash, K. Kumari, K. Prakash, T. Kumar, A transformer-based approach to multimodal
hateful meme classification, in: Working Notes of FIRE 2025 - Forum for Information Retrieval
Evaluation, CEUR, 2025.
[38] T. Kumari, A. Das, A. Sarvaiya, Hateful and ofensive meme detection in multimodal memes dataset
for indo-aryan languages, in: Working Notes of FIRE 2025 - Forum for Information Retrieval
Evaluation, CEUR, 2025.
[39] S. S. Sahu, J. Damor, A transformer-based model for hate speech detection in low resource indian
languages, in: Working Notes of FIRE 2025 - Forum for Information Retrieval Evaluation, CEUR,
2025.
[40] R. B N, S. H L, Towards safer social media: Multimodal hate speech detection in memes across
diverse indian languages, in: Working Notes of FIRE 2025 - Forum for Information Retrieval
Evaluation, CEUR, 2025.
[41] W. Kongqiang, T. Qingli, Hate speech and ofensive content identification in memes in bangla,
hindi and gujarati using auxiliary text supervised learning, in: Working Notes of FIRE 2025
Forum for Information Retrieval Evaluation, CEUR, 2025.
[42] K. Tewari, S. Chanda, A. Namdeo, S. Pal, Savior: Sentiment, sarcasm, abuse, and vulgarity in online
realities (memes), in: Working Notes of FIRE 2025 - Forum for Information Retrieval Evaluation,
CEUR, 2025.
[43] S. Singh, G. Kumar, J. P. Singh, S. K. Rai, K. Goswami, Multilingual hate speech classification in
memes using ocr-extracted text and visual features, in: Working Notes of FIRE 2025 - Forum for
Information Retrieval Evaluation, CEUR, 2025.
[44] A. Malviya, A. Basak, S. Choudhury, Fusionguard: Visual-linguistic representations for multilingual
harmful content detection, in: Working Notes of FIRE 2025 - Forum for Information Retrieval
Evaluation, CEUR, 2025.
[45] P. Dadure, S. Ghosh, Bodo meme classification using multimodal fusion and oversampling in
hasoc-2025, in: Working Notes of FIRE 2025 - Forum for Information Retrieval Evaluation, CEUR,
2025.
[46] M. Rafi, Fast-hasoc 2025: Multimodal and multilingual approaches for hate speech and ofensive
content detection in hindi memes, in: Working Notes of FIRE 2025 - Forum for Information</p>
    </sec>
  </body>
  <back>
    <ref-list>
      <ref id="ref1">
        <mixed-citation>
          [1]
          <string-name>
            <given-names>R.</given-names>
            <surname>Dawkins</surname>
          </string-name>
          , The Selfish Gene, new edition ed., Oxford University Press, Oxford, UK,
          <year>1989</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref2">
        <mixed-citation>
          [2]
          <string-name>
            <given-names>S.</given-names>
            <surname>Pramanick</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.</given-names>
            <surname>Sharma</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Dimitrov</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M. S.</given-names>
            <surname>Akhtar</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Nakov</surname>
          </string-name>
          , T. Chakraborty,
          <article-title>Momenta: A multimodal framework for detecting harmful memes and their targets</article-title>
          ,
          <source>in: Findings of the Association for Computational Linguistics: EMNLP</source>
          <year>2021</year>
          ,
          <year>2021</year>
          , pp.
          <fpage>4439</fpage>
          -
          <lpage>4455</lpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref3">
        <mixed-citation>
          [3]
          <string-name>
            <given-names>N.</given-names>
            <surname>Statt</surname>
          </string-name>
          ,
          <article-title>Youtube is facing a full-scale advertising boycott over hate speech (</article-title>
          <year>2017</year>
          ). URL: https://www.theverge.com/
          <year>2017</year>
          /3/24/15054878/ youtube-advertising
          <article-title>-boycott-hate-speech-extremist-content, accessed: [add access date].</article-title>
        </mixed-citation>
      </ref>
      <ref id="ref4">
        <mixed-citation>
          [4]
          <string-name>
            <given-names>J. S.</given-names>
            <surname>Vedeler</surname>
          </string-name>
          ,
          <string-name>
            <given-names>T.</given-names>
            <surname>Olsen</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Eriksen</surname>
          </string-name>
          ,
          <article-title>Hate speech harms: A social justice discussion of disabled norwegians' experiences</article-title>
          ,
          <source>Disability &amp; Society</source>
          <volume>34</volume>
          (
          <year>2019</year>
          )
          <fpage>368</fpage>
          -
          <lpage>383</lpage>
          . doi:
          <volume>10</volume>
          .1080/09687599.
          <year>2018</year>
          .
          <volume>1436033</volume>
          .
        </mixed-citation>
      </ref>
      <ref id="ref5">
        <mixed-citation>
          [5]
          <string-name>
            <given-names>J.</given-names>
            <surname>McSwiney</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Vaughan</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Heft</surname>
          </string-name>
          ,
          <string-name>
            <surname>M.</surname>
          </string-name>
          <article-title>Hofmann, Sharing the hate? Memes and transnationality in the far right's digital visual culture</article-title>
          , Information,
          <source>Communication &amp; Society</source>
          <volume>24</volume>
          (
          <year>2021</year>
          )
          <fpage>2502</fpage>
          -
          <lpage>2521</lpage>
          . URL: https://www.tandfonline.com/doi/full/10.1080/1369118X.
          <year>2021</year>
          .
          <volume>1961006</volume>
          . doi:
          <volume>10</volume>
          .1080/1369118X.
          <year>2021</year>
          .
          <volume>1961006</volume>
          .
        </mixed-citation>
      </ref>
      <ref id="ref6">
        <mixed-citation>
          [6]
          <string-name>
            <given-names>U. K.</given-names>
            <surname>Schmid</surname>
          </string-name>
          ,
          <article-title>Humorous hate speech on social media: A mixed-methods investigation of users' perceptions and processing of hateful memes</article-title>
          , New Media &amp;
          <string-name>
            <surname>Society</surname>
          </string-name>
          (
          <year>2023</year>
          ). URL: http://journals. sagepub.com/doi/10.1177/14614448231198169. doi:
          <volume>10</volume>
          .1177/14614448231198169.
        </mixed-citation>
      </ref>
      <ref id="ref7">
        <mixed-citation>
          [7]
          <string-name>
            <given-names>R.</given-names>
            <surname>Gomez</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Gibert</surname>
          </string-name>
          ,
          <string-name>
            <given-names>L.</given-names>
            <surname>Gomez</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Karatzas</surname>
          </string-name>
          ,
          <article-title>Exploring hate speech detection in multimodal publications</article-title>
          ,
          <source>in: Proceedings of the IEEE/CVF Winter Conference on Applications of Computer Vision (WACV)</source>
          ,
          <year>2020</year>
          , pp.
          <fpage>1470</fpage>
          -
          <lpage>1478</lpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref8">
        <mixed-citation>
          [8]
          <string-name>
            <given-names>B. O.</given-names>
            <surname>Sabat</surname>
          </string-name>
          ,
          <string-name>
            <given-names>C.</given-names>
            <surname>Canton Ferrer</surname>
          </string-name>
          ,
          <string-name>
            <surname>X.</surname>
          </string-name>
          <article-title>Giro-i Nieto, Hate speech in pixels: Detection of ofensive memes towards automatic moderation</article-title>
          , arXiv preprint arXiv:
          <year>1910</year>
          .
          <volume>02334</volume>
          (
          <year>2019</year>
          ). URL: https://arxiv.org/ abs/
          <year>1910</year>
          .02334.
        </mixed-citation>
      </ref>
      <ref id="ref9">
        <mixed-citation>
          [9]
          <string-name>
            <given-names>S.</given-names>
            <surname>Suryawanshi</surname>
          </string-name>
          ,
          <string-name>
            <given-names>B. R.</given-names>
            <surname>Chakravarthi</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Arcan</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Buitelaar</surname>
          </string-name>
          ,
          <article-title>Multimodal meme dataset (multiof) for identifying ofensive content in image and text</article-title>
          ,
          <source>in: Proceedings of the Second Workshop on Trolling, Aggression and Cyberbullying</source>
          ,
          <year>2020</year>
          , pp.
          <fpage>32</fpage>
          -
          <lpage>41</lpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref10">
        <mixed-citation>
          [10]
          <string-name>
            <given-names>S.</given-names>
            <surname>Pramanick</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Dimitrov</surname>
          </string-name>
          ,
          <string-name>
            <given-names>R.</given-names>
            <surname>Mukherjee</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.</given-names>
            <surname>Sharma</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M. S.</given-names>
            <surname>Akhtar</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Nakov</surname>
          </string-name>
          , T. Chakraborty,
          <article-title>Detecting harmful memes and their targets, in: Findings of the Association for Computational Linguistics: ACL-IJCNLP</article-title>
          <year>2021</year>
          ,
          <year>2021</year>
          , pp.
          <fpage>2783</fpage>
          -
          <lpage>2796</lpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref11">
        <mixed-citation>
          [11]
          <string-name>
            <given-names>S.</given-names>
            <surname>Pramanick</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.</given-names>
            <surname>Sharma</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Dimitrov</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M. S.</given-names>
            <surname>Akhtar</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Nakov</surname>
          </string-name>
          , T. Chakraborty,
          <article-title>Momenta: A multimodal framework for detecting harmful memes and their targets</article-title>
          ,
          <source>in: Findings of the Association for Computational Linguistics: EMNLP</source>
          <year>2021</year>
          ,
          <year>2021</year>
          , pp.
          <fpage>4439</fpage>
          -
          <lpage>4455</lpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref12">
        <mixed-citation>
          [12]
          <string-name>
            <given-names>S.</given-names>
            <surname>Modha</surname>
          </string-name>
          ,
          <string-name>
            <given-names>T.</given-names>
            <surname>Mandl</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Majumder</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Patel</surname>
          </string-name>
          ,
          <source>Overview of the HASOC track at FIRE</source>
          <year>2019</year>
          :
          <article-title>Hate speech and ofensive content identification in Indo-European Languages</article-title>
          , in: Working Notes of FIRE 2019 -
          <article-title>Forum for Information Retrieval Evaluation, Kolkata</article-title>
          , India, Dec.
          <fpage>12</fpage>
          -
          <lpage>15</lpage>
          , volume
          <volume>2517</volume>
          , CEUR-WS.org,
          <year>2019</year>
          , pp.
          <fpage>167</fpage>
          -
          <lpage>190</lpage>
          . URL: http://ceur-ws.
          <source>org/</source>
          Vol-
          <volume>2517</volume>
          /
          <fpage>T3</fpage>
          -1.pdf.
        </mixed-citation>
      </ref>
      <ref id="ref13">
        <mixed-citation>
          [13]
          <string-name>
            <given-names>B. O.</given-names>
            <surname>Sabat</surname>
          </string-name>
          ,
          <string-name>
            <given-names>C. C.</given-names>
            <surname>Ferrer</surname>
          </string-name>
          ,
          <string-name>
            <surname>X. G. i Nieto</surname>
          </string-name>
          ,
          <article-title>Hate speech in pixels: Detection of ofensive memes towards automatic moderation</article-title>
          ,
          <year>2019</year>
          . URL: https://arxiv.org/abs/
          <year>1910</year>
          .02334. arXiv:
          <year>1910</year>
          .02334.
        </mixed-citation>
      </ref>
      <ref id="ref14">
        <mixed-citation>
          [14]
          <string-name>
            <given-names>R.</given-names>
            <surname>Gomez</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Gibert</surname>
          </string-name>
          ,
          <string-name>
            <given-names>L.</given-names>
            <surname>Gomez</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Karatzas</surname>
          </string-name>
          ,
          <article-title>Exploring hate speech detection in multimodal publications</article-title>
          ,
          <year>2019</year>
          . URL: https://arxiv.org/abs/
          <year>1910</year>
          .03814. arXiv:
          <year>1910</year>
          .03814.
        </mixed-citation>
      </ref>
      <ref id="ref15">
        <mixed-citation>
          [15]
          <string-name>
            <given-names>M.</given-names>
            <surname>Chandra</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Pailla</surname>
          </string-name>
          ,
          <string-name>
            <given-names>H.</given-names>
            <surname>Bhatia</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Sanchawala</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Gupta</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Shrivastava</surname>
          </string-name>
          , P. Kumaraguru,
          <article-title>“subverting the jewtocracy”: Online antisemitism detection using multimodal deep learning</article-title>
          ,
          <source>in: Proceedings of the 13th ACM Web Science Conference</source>
          <year>2021</year>
          , WebSci '21,
          <string-name>
            <surname>Association</surname>
          </string-name>
          for Computing Machinery, New York, NY, USA,
          <year>2021</year>
          , p.
          <fpage>148</fpage>
          -
          <lpage>157</lpage>
          . URL: https://doi.org/10.1145/3447535.3462502. doi:
          <volume>10</volume>
          .1145/3447535.3462502.
        </mixed-citation>
      </ref>
      <ref id="ref16">
        <mixed-citation>
          [16]
          <string-name>
            <given-names>S.</given-names>
            <surname>Suryawanshi</surname>
          </string-name>
          ,
          <string-name>
            <given-names>B. R.</given-names>
            <surname>Chakravarthi</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Arcan</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Buitelaar</surname>
          </string-name>
          ,
          <article-title>Multimodal meme dataset (MultiOFF) for identifying ofensive content in image and text</article-title>
          ,
          <source>in: Proceedings of the Second Workshop on Trolling, Aggression and Cyberbullying</source>
          ,
          <string-name>
            <surname>ELRA</surname>
          </string-name>
          , Marseille, France,
          <year>2020</year>
          , pp.
          <fpage>32</fpage>
          -
          <lpage>41</lpage>
          . URL: https: //aclanthology.org/
          <year>2020</year>
          .trac-
          <volume>1</volume>
          .6.
        </mixed-citation>
      </ref>
      <ref id="ref17">
        <mixed-citation>
          [17]
          <string-name>
            <given-names>S.</given-names>
            <surname>Pramanick</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Dimitrov</surname>
          </string-name>
          ,
          <string-name>
            <given-names>R.</given-names>
            <surname>Mukherjee</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.</given-names>
            <surname>Sharma</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M. S.</given-names>
            <surname>Akhtar</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Nakov</surname>
          </string-name>
          , T. Chakraborty,
          <article-title>Detecting harmful memes and their targets</article-title>
          , in: C.
          <string-name>
            <surname>Zong</surname>
            ,
            <given-names>F.</given-names>
          </string-name>
          <string-name>
            <surname>Xia</surname>
            ,
            <given-names>W.</given-names>
          </string-name>
          <string-name>
            <surname>Li</surname>
            ,
            <given-names>R.</given-names>
          </string-name>
          <string-name>
            <surname>Navigli</surname>
          </string-name>
          (Eds.),
          <article-title>Findings of the Association for Computational Linguistics: ACL-IJCNLP 2021</article-title>
          ,
          <article-title>Association for Computational Linguistics</article-title>
          , Online,
          <year>2021</year>
          , pp.
          <fpage>2783</fpage>
          -
          <lpage>2796</lpage>
          . URL: https://aclanthology.org/
          <year>2021</year>
          .findings-acl.
          <volume>246</volume>
          /. doi:
          <volume>10</volume>
          .18653/v1/
          <year>2021</year>
          .findings- acl.246.
        </mixed-citation>
      </ref>
      <ref id="ref18">
        <mixed-citation>
          [18]
          <string-name>
            <given-names>D.</given-names>
            <surname>Kiela</surname>
          </string-name>
          ,
          <string-name>
            <given-names>H.</given-names>
            <surname>Firooz</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Mohan</surname>
          </string-name>
          ,
          <string-name>
            <given-names>V.</given-names>
            <surname>Goswami</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Singh</surname>
          </string-name>
          ,
          <string-name>
            <given-names>C. A.</given-names>
            <surname>Fitzpatrick</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Bull</surname>
          </string-name>
          , G. Lipstein,
          <string-name>
            <given-names>T.</given-names>
            <surname>Nelli</surname>
          </string-name>
          ,
          <string-name>
            <given-names>R.</given-names>
            <surname>Zhu</surname>
          </string-name>
          , et al.,
          <article-title>The hateful memes challenge: competition report</article-title>
          , in: NeurIPS 2020 Competition and Demonstration Track,
          <string-name>
            <surname>PMLR</surname>
          </string-name>
          ,
          <year>2021</year>
          , pp.
          <fpage>344</fpage>
          -
          <lpage>360</lpage>
          . URL: https://proceedings.mlr.press/v133/ kiela21a.html.
        </mixed-citation>
      </ref>
      <ref id="ref19">
        <mixed-citation>
          [19]
          <string-name>
            <given-names>E.</given-names>
            <surname>Fersini</surname>
          </string-name>
          ,
          <string-name>
            <given-names>F.</given-names>
            <surname>Gasparini</surname>
          </string-name>
          ,
          <string-name>
            <given-names>G.</given-names>
            <surname>Rizzi</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Saibene</surname>
          </string-name>
          ,
          <string-name>
            <given-names>B.</given-names>
            <surname>Chulvi</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Rosso</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Lees</surname>
          </string-name>
          , J. Sorensen, SemEval
          <article-title>-2022 Task 5: Multimedia automatic misogyny identification</article-title>
          ,
          <source>in: Proceedings of the 16th International Workshop onSemantic Evaluation (SemEval)</source>
          , ACL, Ann Arbor, Michigan,
          <year>2022</year>
          . doi:
          <volume>10</volume>
          .18653/v1/
          <year>2022</year>
          .semeval-
          <volume>1</volume>
          .
          <fpage>74</fpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref20">
        <mixed-citation>
          [20]
          <string-name>
            <given-names>L.</given-names>
            <surname>Plaza</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Carrillo-de Albornoz</surname>
          </string-name>
          , E. Amigó,
          <string-name>
            <given-names>J.</given-names>
            <surname>Gonzalo</surname>
          </string-name>
          ,
          <string-name>
            <given-names>R.</given-names>
            <surname>Morante</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Rosso</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Spina</surname>
          </string-name>
          ,
          <string-name>
            <given-names>B.</given-names>
            <surname>Chulvi</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Maeso</surname>
          </string-name>
          ,
          <string-name>
            <given-names>V.</given-names>
            <surname>Ruiz</surname>
          </string-name>
          ,
          <year>Exist 2024</year>
          :
          <article-title>sexism identification in social networks and memes</article-title>
          , in: N.
          <string-name>
            <surname>Goharian</surname>
            ,
            <given-names>N.</given-names>
          </string-name>
          <string-name>
            <surname>Tonellotto</surname>
            ,
            <given-names>Y.</given-names>
          </string-name>
          <string-name>
            <surname>He</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          <string-name>
            <surname>Lipani</surname>
            ,
            <given-names>G.</given-names>
          </string-name>
          <string-name>
            <surname>McDonald</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          <string-name>
            <surname>Macdonald</surname>
          </string-name>
          , I. Ounis (Eds.),
          <source>Advances in Information Retrieval</source>
          , Springer Nature Switzerland, Cham,
          <year>2024</year>
          , pp.
          <fpage>498</fpage>
          -
          <lpage>504</lpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref21">
        <mixed-citation>
          [21]
          <string-name>
            <surname>D. S. M. Pandiani</surname>
            ,
            <given-names>E. T. K.</given-names>
          </string-name>
          <string-name>
            <surname>Sang</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          <string-name>
            <surname>Ceolin</surname>
          </string-name>
          ,
          <article-title>Toxic memes: A survey of computational perspectives on the detection and explanation of meme toxicities</article-title>
          ,
          <source>arXiv preprint arXiv:2406.07353</source>
          (
          <year>2024</year>
          ).
        </mixed-citation>
      </ref>
      <ref id="ref22">
        <mixed-citation>
          [22]
          <string-name>
            <given-names>M. R.</given-names>
            <surname>Karim</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S. K.</given-names>
            <surname>Dey</surname>
          </string-name>
          ,
          <string-name>
            <given-names>T.</given-names>
            <surname>Islam</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Shajalal</surname>
          </string-name>
          ,
          <string-name>
            <given-names>B. R.</given-names>
            <surname>Chakravarthi</surname>
          </string-name>
          ,
          <article-title>Multimodal hate speech detection from bengali memes</article-title>
          and texts,
          <year>2022</year>
          . URL: https://arxiv.org/abs/2204.10196. arXiv:
          <volume>2204</volume>
          .
          <fpage>10196</fpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref23">
        <mixed-citation>
          [23]
          <string-name>
            <given-names>E.</given-names>
            <surname>Hossain</surname>
          </string-name>
          ,
          <string-name>
            <given-names>O.</given-names>
            <surname>Sharif</surname>
          </string-name>
          ,
          <string-name>
            <surname>M.</surname>
          </string-name>
          <article-title>M. Hoque, MUTE: A multimodal dataset for detecting hateful memes</article-title>
          , in: Y. Hanqi,
          <string-name>
            <given-names>Y.</given-names>
            <surname>Zonghan</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.</given-names>
            <surname>Ruder</surname>
          </string-name>
          , W. Xiaojun (Eds.),
          <source>Proceedings of the 2nd Conference of the Asia-Pacific Chapter of the Association for Computational Linguistics and the 12th International Joint Conference on Natural Language Processing: Student Research Workshop</source>
          , Association for Computational Linguistics, Online,
          <year>2022</year>
          , pp.
          <fpage>32</fpage>
          -
          <lpage>39</lpage>
          . URL: https://aclanthology.org/
          <year>2022</year>
          .aacl-srw.5/. doi:
          <volume>10</volume>
          .18653/v1/
          <year>2022</year>
          .aacl- srw.5.
        </mixed-citation>
      </ref>
      <ref id="ref24">
        <mixed-citation>
          [24]
          <string-name>
            <surname>M. Das</surname>
            ,
            <given-names>A. Mukherjee,</given-names>
          </string-name>
          <article-title>BanglaAbuseMeme: A dataset for Bengali abusive meme classification</article-title>
          , in: H.
          <string-name>
            <surname>Bouamor</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          <string-name>
            <surname>Pino</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          Bali (Eds.),
          <source>Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing</source>
          , Association for Computational Linguistics, Singapore,
          <year>2023</year>
          , pp.
          <fpage>15498</fpage>
          -
          <lpage>15512</lpage>
          . URL: https://aclanthology.org/
          <year>2023</year>
          .emnlp-main.
          <volume>959</volume>
          /. doi:
          <volume>10</volume>
          .18653/v1/
          <year>2023</year>
          . emnlp- main.959.
        </mixed-citation>
      </ref>
      <ref id="ref25">
        <mixed-citation>
          [25]
          <string-name>
            <given-names>H.</given-names>
            <surname>Kirk</surname>
          </string-name>
          ,
          <string-name>
            <given-names>Y.</given-names>
            <surname>Jun</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Rauba</surname>
          </string-name>
          , G. Wachtel,
          <string-name>
            <given-names>R.</given-names>
            <surname>Li</surname>
          </string-name>
          ,
          <string-name>
            <given-names>X.</given-names>
            <surname>Bai</surname>
          </string-name>
          ,
          <string-name>
            <given-names>N.</given-names>
            <surname>Broestl</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Dof-Sotta</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Shtedritski</surname>
          </string-name>
          ,
          <string-name>
            <given-names>Y. M.</given-names>
            <surname>Asano</surname>
          </string-name>
          ,
          <article-title>Memes in the wild: Assessing the generalizability of the hateful memes challenge dataset</article-title>
          , in: A.
          <string-name>
            <surname>Mostafazadeh Davani</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          <string-name>
            <surname>Kiela</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          <string-name>
            <surname>Lambert</surname>
            ,
            <given-names>B.</given-names>
          </string-name>
          <string-name>
            <surname>Vidgen</surname>
            ,
            <given-names>V.</given-names>
          </string-name>
          <string-name>
            <surname>Prabhakaran</surname>
            ,
            <given-names>Z.</given-names>
          </string-name>
          Waseem (Eds.),
          <source>Proceedings of the 5th Workshop on Online Abuse and Harms (WOAH</source>
          <year>2021</year>
          ), ACL, Online,
          <year>2021</year>
          , pp.
          <fpage>26</fpage>
          -
          <lpage>35</lpage>
          . URL: https://aclanthology.org/
          <year>2021</year>
          .woah-
          <volume>1</volume>
          .4. doi:
          <volume>10</volume>
          .18653/v1/
          <year>2021</year>
          .woah-
          <volume>1</volume>
          .4.
        </mixed-citation>
      </ref>
      <ref id="ref26">
        <mixed-citation>
          [26]
          <string-name>
            <given-names>G.</given-names>
            <surname>Kumari</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Bandyopadhyay</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Ekbal</surname>
          </string-name>
          ,
          <article-title>Emofmeme: identifying ofensive memes by leveraging underlying emotions</article-title>
          ,
          <source>Multimedia Tools Appl</source>
          .
          <volume>82</volume>
          (
          <year>2023</year>
          )
          <fpage>45061</fpage>
          -
          <lpage>45096</lpage>
          . URL: https://doi.org/10.1007/ s11042-023-14807-1. doi:
          <volume>10</volume>
          .1007/s11042- 023- 14807- 1.
        </mixed-citation>
      </ref>
      <ref id="ref27">
        <mixed-citation>
          [27]
          <string-name>
            <given-names>K.</given-names>
            <surname>Rajput</surname>
          </string-name>
          ,
          <string-name>
            <given-names>R.</given-names>
            <surname>Kapoor</surname>
          </string-name>
          ,
          <string-name>
            <given-names>K.</given-names>
            <surname>Rai</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Kaur</surname>
          </string-name>
          ,
          <article-title>Hate me not: Detecting hate inducing memes in code switched languages</article-title>
          ,
          <year>2022</year>
          . URL: https://arxiv.org/abs/2204.11356. arXiv:
          <volume>2204</volume>
          .
          <fpage>11356</fpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref28">
        <mixed-citation>
          [28]
          <string-name>
            <given-names>K.</given-names>
            <surname>Maity</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Jha</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.</given-names>
            <surname>Saha</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Bhattacharyya</surname>
          </string-name>
          ,
          <article-title>A multitask framework for sentiment, emotion and sarcasm aware cyberbullying detection from multi-modal code-mixed memes</article-title>
          ,
          <source>in: Proceedings of the 45th International ACM SIGIR Conference on Research and Development in Information Retrieval</source>
          , SIGIR '22,
          <string-name>
            <surname>Association</surname>
          </string-name>
          for Computing Machinery, New York, NY, USA,
          <year>2022</year>
          , p.
          <fpage>1739</fpage>
          -
          <lpage>1749</lpage>
          . URL: https://doi.org/10.1145/3477495.3531925. doi:
          <volume>10</volume>
          .1145/3477495.3531925.
        </mixed-citation>
      </ref>
      <ref id="ref29">
        <mixed-citation>
          [29]
          <string-name>
            <given-names>P.</given-names>
            <surname>Jha</surname>
          </string-name>
          ,
          <string-name>
            <given-names>K.</given-names>
            <surname>Maity</surname>
          </string-name>
          ,
          <string-name>
            <given-names>R.</given-names>
            <surname>Jain</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Verma</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.</given-names>
            <surname>Saha</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Bhattacharyya</surname>
          </string-name>
          ,
          <article-title>Meme-ingful analysis: Enhanced understanding of cyberbullying in memes through multimodal explanations</article-title>
          , in: Y. Graham, M. Purver (Eds.),
          <source>Proceedings of the 18th Conference of the European Chapter of the Association for Computational Linguistics (Volume</source>
          <volume>1</volume>
          :
          <string-name>
            <surname>Long</surname>
            <given-names>Papers)</given-names>
          </string-name>
          ,
          <source>Association for Computational Linguistics, St. Julian's, Malta</source>
          ,
          <year>2024</year>
          , pp.
          <fpage>930</fpage>
          -
          <lpage>943</lpage>
          . URL: https://aclanthology.org/
          <year>2024</year>
          .
          <article-title>eacl-long</article-title>
          .
          <volume>56</volume>
          /. doi:
          <volume>10</volume>
          . 18653/v1/
          <year>2024</year>
          .eacl- long.56.
        </mixed-citation>
      </ref>
    </ref-list>
  </back>
</article>