<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.0 20120330//EN" "JATS-archivearticle1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta />
    <article-meta>
      <title-group>
        <article-title>An Extended Overview of the CLEF 2020 ChEMU Lab: Information Extraction of Chemical Reactions from Patents</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <string-name>Jiayuan He</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Dat Quoc Nguyen</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Saber A. Akhondi</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Christian Druckenbrodt</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Camilo Thorne</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ralph Hoessel</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Zubair Afzal</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Zenan Zhai</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Biaoyan Fang</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Hiyori Yoshikawa</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ameer Albahem</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Jingqi Wang</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Yuankai Ren</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Zhi Zhang</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Yaoyun Zhang</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Mai Hoang Dao</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Pedro Ruas</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Andre Lamurias</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Francisco M Couto</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Jenny Copara</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Nona Naderi</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Julien Knafou</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Patrick Ruch</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Douglas Teodoro</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Daniel Lowe</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>John May eld</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Abdullatif Koksal</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Hilal Donmez</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Elif O zk r ml</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Arzucan O zgur</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Darshini Mahendran</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Gabrielle Gurdin</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Nastassja Lewinski</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Christina Tang</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Bridget T. McInnes</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Malarkodi C.S.</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Pattabhi Rk Rao.</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Sobha Lalitha Devi</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Lawrence Cavedon</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Trevor Cohn</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Timothy Baldwin</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Karin Verspoor</string-name>
          <email>karin.verspoorg@unimelb.edu.au</email>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <aff id="aff0">
          <label>0</label>
          <institution>The University of Melbourne</institution>
          ,
          <addr-line>Melbourne</addr-line>
          ,
          <country country="AU">Australia</country>
        </aff>
      </contrib-group>
    </article-meta>
  </front>
  <body>
    <sec id="sec-1">
      <title>-</title>
      <p>Abstract. The discovery of new chemical compounds is perceived as a
key driver of the chemistry industry and many other economic sectors.
The information about the new discoveries are usually disclosed in
scienti c literature and in particular, in chemical patents, since patents are
often the rst venues where the new chemical compounds are publicized.
Despite the signi cance of the information provided in chemical patents,
extracting the information from patents is costly due to the large volume
of existing patents and its drastic expansion rate. The
Cheminformatics Elsevier Melbourne University (ChEMU) evaluation lab 2020, part
of the Conference and Labs of the Evaluation Forum 2020 (CLEF2020),
provides a platform to advance the state-of-the-arts in automatic
information extraction systems over chemical patents. In particular, we
focus on extracting synthesis process of new chemical compounds from
chemical patents. Using the ChEMU corpus of 1500 \snippets" (text
segments) sampled from 170 patent documents and annotated by chemical
experts, we de ned two key information extraction tasks. Task 1 targets
at chemical named entity recognition, i.e., the identi cation of chemical
compounds and their speci c roles in chemical reactions. Task 2 targets
at event extraction, i.e., the identi cation of reaction steps, relating the
chemical compounds involved in a chemical reaction. In this paper, we
provide an overview of our ChEMU2020 lab. Herein, we describe the
resources created for the two tasks, the evaluation methodology adopted,
and participants results. We also provide a brief summary of the
methods employed by participants of this lab and the results obtained across
46 runs from 11 teams, nding that several submissions achieve
substantially better results than the baseline methods prepared by the
organizers.
1</p>
    </sec>
    <sec id="sec-2">
      <title>Introduction</title>
      <p>
        Chemical patents represent as an indispensable source information about new
discoveries in chemistry. They are usually the rst venues where new chemical
compounds are disclosed [
        <xref ref-type="bibr" rid="ref7">7,40</xref>
        ] and can lead general scienti c literature (e.g.,
journal articles) by up-to 3 years. In addition, chemical patents usually
contain much more comprehensive information about the synthesis process of new
chemical compounds including their reaction steps and experimental conditions
for compound synthesis and mode of action. These details are crucial for the
understanding of compound prior art, and provide a means for novelty checking
and validation [
        <xref ref-type="bibr" rid="ref5 ref6">5,6</xref>
        ].
      </p>
      <p>
        Although the information in chemical patents are of signi cant research and
commercial value, extracting such information is nontrivial, since the large
volCopyright c 2020 for this paper by its authors. Use permitted under Creative
Commons License Attribution 4.0 International (CC BY 4.0). CLEF 2020, 22-25
September 2020, Thessaloniki, Greece.
ume of existing patents and its drastic expansion rate has made manual
annotation costly and time-consuming [
        <xref ref-type="bibr" rid="ref29">29</xref>
        ]. Natural language processing (NLP) refer
to techniques that allow computers to automatically analyze and process
natural unstructured language data, and it has enjoyed great success over the past
decades [
        <xref ref-type="bibr" rid="ref30">30,44</xref>
        ]. In light of this, researchers have been actively exploring the
possible application of NLP techniques to patent text mining, so as to alleviate the
time-consuming e orts of manual annotation by chemical experts and scale the
information extraction process over chemical patents.
      </p>
      <p>
        The ChEMU (Cheminformatics Elsevier Melbourne University) lab aims to
provide a platform for worldwide experts in both NLP and chemistry to
develop automated information extraction methods over chemical patents, and
to advance the state-of-the-arts in this area. As a rst running of ChEMU, our
ChEMU2020 lab focuses on extraction of chemical reactions from patents [
        <xref ref-type="bibr" rid="ref14 ref32">32,14</xref>
        ].
Speci cally, we provided two information extraction tasks that are crucial steps
for chemical reaction extraction. The rst task, named entity recognition,
requires the identi cation of essential elements of chemical reactions, such as
chemical compounds involved, conditions at which reactions are carried out,
and yields of reactions. We go beyond identifying named entities and also
require identi cation of their speci c roles in chemical reactions. The second task,
event extraction, requires the identi cation of speci c event steps that are
performed in a chemical reaction.
      </p>
      <p>In collaboration with chemical domain experts, we have prepared a
highquality annotated data set of 1,500 segments of chemical patent texts speci cally
targeting these two tasks. The 1,500 segments are sampled from 170 chemical
patents, and each segment contains a meaningful chemical reaction. Annotations
including entities and event steps are rstly prepared by three chemical experts
and then merged to gold-standards.</p>
      <p>The ChEMU2020 lab has received considerable interest, attracting 37
registrants from 13 countries including Portugal, Switzerland, Germany, India, Japan,
United States, China, and United Kingdom. Speci cally, we received 26 runs (1
post-evaluation submission) from 11 teams in Task 1, 10 runs from 5 teams in
Task 2, and 10 runs from 4 teams in the task of end-to-end systems (a pipeline
combining Task 1 and 2), respectively. Several teams achieved exciting results,
outperforming baseline models signi cantly. In particular, submissions from a
team from the company Melax Technologies (from Houston, TX, USA) ranked
rst in all 3 tasks.</p>
      <p>
        The rest of the paper is structured as follows. We rst introduce the corpus
we created for use in the lab in Sect. 2. Then we give an overview of the tasks
and tracks in Sect. 3, and discuss the evaluation framework used in the lab in
Sect. 4. We present the overall evaluation results in Sect. 5 and introduce the
participants' approaches in Sect. 6, comparing them in Sect. 7. Conclusions are
presented in Sect. 8. Note that this paper is an extension of our previous overview
paper [
        <xref ref-type="bibr" rid="ref14">14</xref>
        ] and thereby Sect. 2 to 4 here are repeated from that paper; our focus
is to provide additional methodological detail.
      </p>
    </sec>
    <sec id="sec-3">
      <title>The ChEMU Chemical Reaction Corpus</title>
      <p>The annotated corpus prepared for the ChEMU shared task consists of 1,500
patent snippets (text segments) that were sampled from 170 English document
patents from the European Patent O ce and the United States Patent and
Trademark O ce. Each snippet contains a meaningful description of a chemical
reaction [47].</p>
      <p>
        The corpus was based on information captured in the Reaxys R database.1
This resource contains details of chemical reactions identi ed through a mostly
manual process of extracting key reaction details from sources including patents
and scienti c publications, dubbed \excerption" [
        <xref ref-type="bibr" rid="ref20">20</xref>
        ].
2.1
      </p>
      <sec id="sec-3-1">
        <title>Annotation Process</title>
        <p>
          To prepare the gold-standard annotations for the extracted patent snippets,
multiple domain experts with rich expert knowledge in chemistry were invited to
assist with corpus annotation. A silver-standard annotation set was rst generated
by mapping the records from the Reaxys database back to the source patents
from which the records were originally extracted. This was done by scanning
the patent texts for mentions of relevant entities. Since the original records are
only linked to the IDs of source patents and do not provide the precise locations
of excerpted entities or event steps, these annotations needed to be manually
reviewed to produce higher-quality annotations. Two domain experts manually
and independently reviewed all patent snippets, correcting location information
of the annotations in silver-standard annotations and adding more annotations.
Their annotations were then evaluated by measuring their inter-annotator
agreement (IAA) [
          <xref ref-type="bibr" rid="ref8">8</xref>
          ], and thereafter merged by a third domain expert who acted as
an adjudicator, to resolve di erences. More details about the quality evaluation
over the annotations and the harmonization process will be provided in a more
in-depth paper to follow.
        </p>
        <p>We present an example of a patent snippet in Fig. 1. This snippet describes
the synthesis of a particular chemical compound, named N-((5-(hydrazinecarbonyl)
pyridin-2-yl)methyl)-1-methyl-N-phenylpiperidine-4-carboxamide. The synthesis
process consists of an ordered sequence of reaction steps: (1) dissolving the
chemical compound synthesized in step 3 and hydrazine monohydrate in ethanol; (2)
heating the solution under re ux; (3) cooling the solution to room temperature;
(4) concentrating the cooled mixture under reduced pressure; (5) puri cation of
the concentrate by column chromatography; and (6) concentration of the puri ed
product to get the title compound.</p>
        <p>This shared task aims at extraction of chemical reactions from chemical
patents, e.g., extracting the above synthesis steps given the patent snippet in
Fig. 1. To achieve this goal, it is crucial for us to rst identify the entities that
are involved in these reaction steps (e.g., hydrazine monohydrate and ethanol)
1 https://www.reaxys.com Reaxys R Copyright c 2020 Elsevier Limited except
certain content provided by third parties. Reaxys is a trademark of Elsevier Limited.</p>
        <p>An example snippet
[Step 4] Synthesis of
N-((5-(hydrazinecarbonyl)pyridin-2-yl)methyl)-1-methylN-phenylpiperidine-4-carboxamide Methyl
6-((1-methyl-N-phenylpiperidine4-carboxamido)methyl)nicotinate (0.120 g, 0.327 mmol), synthesized in step 3,
and hydrazine monohydrate (0.079 mL, 1.633 mmol) were dissolved in ethanol
(10 mL) at room temperature, and the solution was heated under re ux for
12 hours, and then cooled to room temperature to terminate the reaction.
The reaction mixture was concentrated under reduced pressure to remove the
solvent, and the concentrate was puri ed by column chromatography (SiO2, 4
g cartridge; methanol/dichloromethane = from 5% to 30%) and concentrated
to give the title compound (0.115 g, 95.8%) as a foam solid.
and then determine the relations between the involved entities (e.g., hydrazine
monohydrate is dissolved in ethanol). Thus, our annotation process consists of
two steps: named entity annotations and relation annotations. Next, we describe
the two steps of annotations in Sect. 2.2 and Sect. 2.3, respectively.
2.2</p>
      </sec>
      <sec id="sec-3-2">
        <title>Named Entity Annotations</title>
        <p>Four categories of entities are annotated over the corpus: (1) chemical
compounds that are involved in a chemical reaction; (2) conditions under which a
chemical reaction is carried out; (3) yields obtained for the nal chemical
product; and (4) example labels that are associated with reaction speci cations.</p>
        <p>Ten labels are further de ned under the above four categories. We de ne
ve di erent roles that a chemical compound can play within a chemical
reaction, corresponding to ve labels under this category: STARTING MATERIAL,
REAGENT CATALYST, REACTION PRODUCT, SOLVENT, and OTHER
COMPOUND. For example, the chemical compound \ethanol" in Fig. 1 must
be annotated with the label \SOLVENT".</p>
        <p>
          We also de ne two labels under the category of conditions: TIME and
TEMPERATURE; and two labels under the category of yields: YIELD PERCENT
and YIELD OTHER. The de nitions of all resultant labels are summarized in
Table 1. Interested readers may nd more information about the labels in [
          <xref ref-type="bibr" rid="ref32">32</xref>
          ] and
examples of named entity annotations in the Task 1|NER annotation
guidelines [45].
2.3
        </p>
      </sec>
      <sec id="sec-3-3">
        <title>Relation Annotations</title>
        <p>A chemical reaction step typically involves an action and chemical compound(s)
on which the action takes e ect. We therefore treat the extraction of a reaction
step as a two-stage task: (1) identi cation of a trigger word that indicates a
chemical reaction step; and (2) identi cation of the relation between a trigger word
and chemical compound(s) that is(are) linked to the trigger word. In addition,
we observe that it is also crucial for us to link an action to the conditions under
which the action is carried out, and resultant yields from the action, in order to
fully quantify a reaction step. Thus, annotations in this step are performed to
identify the relations between actions (trigger words) and all arguments that are
involved in the reaction steps, i.e., chemical compounds, conditions, and yields.
REACTION PRODUCT
SOLVENT
OTHER COMPOUND
TIME
TEMPERATURE
YIELD PERCENT
YIELD OTHER</p>
        <p>EXAMPLE LABEL
Relation Annotations</p>
        <p>WORKUP
Arg1
ArgM</p>
        <p>De nition
A substance that is consumed in the course of a
chemical reaction providing atoms to products is
considered as starting material.</p>
        <p>A reagent is a compound added to a system to cause
or help with a chemical reaction.</p>
        <p>A product is a substance that is formed during a
chemical reaction.</p>
        <p>A solvent is a chemical entity that dissolves a solute
resulting in a solution.</p>
        <p>Other chemical compounds that are not the
products, starting materials, reagents, catalysts and
solvents.</p>
        <p>The reaction time of the reaction.</p>
        <p>The temperature at which the reaction was carried
out.</p>
        <p>Yield given in percent values.</p>
        <p>Yields provided in other units than %.</p>
        <p>A label associated with a reaction speci cation.</p>
        <p>An event step which is a manipulation required to
isolate and purify the product of a chemical reaction.</p>
        <p>An event within which starting materials are
converted into the product.</p>
        <p>The relation between an event trigger word and a
chemical compound.</p>
        <p>The relation between an event trigger word and a
temperature, time, or yield entity.</p>
        <p>O sets</p>
        <p>
          We de ne two types of trigger words: WORKUP which refers to an event
step where a chemical compound is isolated/puri ed, and REACTION STEP
which refers to an event step that is involved in the conversion from a starting
material to an end product. When labelling event arguments, we adapt semantic
argument role labels Arg1 and ArgM from the Proposition Bank [
          <xref ref-type="bibr" rid="ref33">33</xref>
          ] to label
the relations between the trigger words and other arguments. Speci cally, the
label Arg1 refers to the relation between an event trigger word and a chemical
compound. Here, Arg1 represents argument roles of being causally a ected by
another participant in the event [
          <xref ref-type="bibr" rid="ref16">16</xref>
          ]. ArgM represents adjunct roles with respect
to an event, used to label the relation between a trigger word and a temperature,
time or yield entity. The de nitions of trigger word types and relation types are
summarized in Table 1. Detailed annotation guidelines for relation annotation
are available online [45].
2.4
        </p>
      </sec>
      <sec id="sec-3-4">
        <title>Snippet Annotation Format</title>
        <p>The gold-standard annotations for the data set were delivered in the BRAT
stando format [42]. For each snippet, two les were delivered: a text le (.txt)
containing the original texts in the snippet, and a paired annotation le (.ann)
containing all the annotations that have been made for that text, including
entities, trigger words, and event steps. Continuing with the above snippet example,
we present the formatted annotations for the highlighted sentence in Tables 2
and 3. For ease of presentation, we show the annotated named entities and
trigger words in Table 2 and the annotated event steps in Table 3. Speci cally, two
entities (i.e., T1 and T2) and one trigger word are included in Table 2, and two
event steps are included in Table 3.
We randomly partitioned the whole data set into three splits for training,
development and test purposes, with a ratio of 0.6/0.15/0.25. The training and
development sets were released to participants for model development. Note that
participants are allowed to use the combination of training and development sets
and to use their own partitions to build models. The test set is withheld for use
in the formal evaluation. The statistics of the three splits including their
number of snippets, total number of sentences, and number of words per snippet, are
summarized in Table 4.</p>
        <p>To ensure a fair split of data as much as possible, we conduct two statistical
tests on the resultant train/dev/test splits. In the rst test, we compare the
distributions of entity labels (ten classes of entities in Task 1 and two classes
of trigger words in Task 2) within train/dev/test sets, to make sure that the
three sets of snippets have similar distributions over labels. The distributions
are summarize in Table 5, where each cell represents the proportion (e.g., 0.038)
of an entity label (e.g., EXAMPLE LABEL) in the gold annotations of a data
split (e.g., Train). The results in Table 5 con rm that the label distributions in
the three splits are similar. Only some slight uctuations (6 0:004) across the
three splits are observed for each label.</p>
        <p>
          We further compare the International Patent Classi cation (IPC) [
          <xref ref-type="bibr" rid="ref3">3</xref>
          ]
distributions of the training, development and test sets. The IPC information of each
patent snippet re ects the application category of the original patent. For
example, the IPC code \A61K" represents the category of patents that are for
preparations for medical, dental, or toilet purposes. Patents with di erent IPCs
may be written in di erent ways and may di er in the vocabulary. Thus, they
may di er in their linguistic characteristics. For each patent snippet, we extract
the primary IPC of its corresponding source patent, and summarize the IPC
distributions of the snippets in train/dev/test sets in Table 6.
3
        </p>
      </sec>
    </sec>
    <sec id="sec-4">
      <title>The Tasks</title>
      <p>We provide two tasks in ChEMU lab: Task 1|Named Entity Recognition (NER),
and Task 2|Event Extraction (EE). We also host a third track where
participants can work on building end-to-end systems addressing both tasks jointly.
In order to understand and extract a chemical reaction from natural language
texts, the rst essential step is to identify the entities that are involved in the
chemical reaction. The rst task aims to accomplish this step by identifying
the ten types of entities described in Sect. 2.2. The task requires the detection
of the entity names in patent snippets and the assignment of correct labels
to the detected entities (see Table 1). For example, given a detected chemical
compound, the task requires the identi cation of both its text span and its
speci c type according to the role in which it plays within a chemical reaction
description.
3.2</p>
      <sec id="sec-4-1">
        <title>Task 2: Event Extraction</title>
        <p>A chemical reaction usually consists of an ordered sequence of event steps that
transforms a starting product to an end product, such as the ve reaction steps
in the synthesis process of the chemical compound described in the example
in Figure 1. The event extraction task (Task 2) targets identifying these event
steps.</p>
        <p>
          Similarly to conventional event extraction problems [
          <xref ref-type="bibr" rid="ref17">17</xref>
          ], Task 2 involves three
subtasks: event trigger word detection, event typing and argument prediction.
First, it requires the detection of event trigger words and assignment of correct
labels for the trigger words. Second, it requires the determination of argument
entities that are associated with the trigger words, i.e., which entities identi ed
in Task 1 participate in event or reaction steps. This is done by labelling the
connections between event trigger words and their arguments. Given an event
trigger word e and a set S of arguments that participate in e, Task 2 requires the
creation of jSj relation entries connecting e to an argument entity in S. Here, jSj
represents the cardinality of the set S. Finally, Task 2 requires the assignment
of correct relation type labels (Arg1 or ArgM) to each of the detected relations.
        </p>
        <p>In the track for Task 2, the gold standard entities in snippets are assumed
to be known input. While in a real-world use of an event extraction system,
gold standard entities would not typically be available, this framework allowed
participants to focus on event extraction in isolation of the NER task.
3.3</p>
      </sec>
      <sec id="sec-4-2">
        <title>Task 3: End-to-End Systems</title>
        <p>We also hosted a third track which allows participants to develop end-to-end
systems that address both tasks simultaneously, i.e., the extraction of reaction
events including their constituent entities directly from chemical patent snippets.
This is a more realistic scenario for an event extraction system to be applied for
large-scale annotation of events.</p>
        <p>In the testing stage, participants in this track were provided only with the
text of a patent, and were required to identify the named entities de ned in
Table 1, the trigger words de ned in Sect. 3.2, and the event steps involving the
entities, that is, the reaction steps. Proposed models in this track were evaluated
against the events that they predict for the test snippets, which is the same as
in Task 2. However, a major di erence between this track and Task 2 is that
the gold named entities were not provided but rather had to be predicted by the
systems.
3.4</p>
      </sec>
      <sec id="sec-4-3">
        <title>Track overview</title>
        <p>We illustrate the work ows of the three tracks in Fig. 2 using as example the
sentence highlighted in Fig 1. In Task 1|NER|, participants need to identify
entities that de ned in Table 1, e.g., the text span \ethanol" is identi ed as
\SOLVENT". In Task 2|EE|, participants are provided with the three gold
standard entities in the sentence. They are required to rstly identify the trigger
words and their types (e.g., the text span \dissolved" is identi ed as
\REACTION STEP") and then identify the relations between the trigger words and the
provided entities (e.g., a directed link from \dissolved" to \ethanol" is added and
labeled as \ARG1"). In the track of end-to-end systems, participants are only
provided with the original text. They are required to identify both the entities
and the trigger words, and predict the event steps directly from the text.
Training stage. In the training stage, the training and development data sets
were released to all participants for model development. To accommodate the
needs of participants in di erent tracks, two di erent versions of training data,
namely Data-NER and Data-EE, were provided. Data-NER was prepared for
participants in Task 1, where the gold-standard entities de ned in Table 1 were
included. Data-EE was prepared for Tasks 2 and 3, where both the gold-standard
entities, annotated trigger words and entity relations were included.
Testing stage. Since the gold-standard entities need to be provided to
participants in Task 2, the testing stage of Task 2 was delayed until after the testing
of Tasks 1 and 3 are completed, in order to prevent any leakage of information.
Therefore, the testing stage consists of two phases. In the rst phase, the text
(.txt) les of all test snippets were released. Participants in Task 1 are required
to use the released patent texts to predict the entities as de ned in Table 1.
Participants in Task 3 were required to also predict the trigger words and entity
relations de ned in Sect. 3.2. In the second phase, the gold-standard entities of
all test snippets were released. Participants in Task 2 can use the released
goldstandard entities, along with the text les released in the rst phase, to predict
the event steps in test snippets.</p>
        <p>Submission website. A submission website has been developed, which allows
participants to submit their runs during the testing stage.2 In addition, the
website o ers several important functions to facilitate organizing the lab.</p>
        <p>
          First, it hosts the download links for the training, development, and test
data sets so that participants can access the data sets conveniently. Second, it
allows participants to test the performance (against the development set) of their
models before the testing stage starts, which also o ers a chance for participants
to familiarize themselves with the evaluation tool BRATEval [
          <xref ref-type="bibr" rid="ref1">1</xref>
          ] (detailed in
Sect. 4). The website also hosts a private leaderboard for each team that ranks
all runs submitted by each team, and a public leaderboard that ranks all runs
that have been made public by teams.
4
        </p>
      </sec>
    </sec>
    <sec id="sec-5">
      <title>Evaluation Framework</title>
      <p>In this section, we describe the evaluation framework of the ChEMU lab. We
introduce three baseline algorithms for Task 1, Task 2, and end-to-end systems,
respectively.
4.1</p>
      <sec id="sec-5-1">
        <title>Evaluation Methods</title>
        <p>
          We use BRATEval [
          <xref ref-type="bibr" rid="ref1">1</xref>
          ] to evaluate all the runs that we receive. Three metrics are
used to evaluate the performance of all the submissions for Task 1: Precision,
Recall, and F1-score. Speci cally, given a predicted entity and a ground-truth
entity, we treat the two entities as a match if (1) the types associated with the
two entities match; and (2) their text spans match. The overall Precision, Recall,
and F1-score are computed by micro-averaging all instances (entities).
        </p>
        <p>In addition, we exploit two di erent matching criteria, exact-match and
relaxed-match, when comparing the texts spans of two entities. Here, the
exactmatch criterion means that we consider that the text span of an entity matches
with that of another entity if both the starting and the end o sets of their spans
match. The relaxed-match criterion means that we consider that the text span
of one entity matches with that of another entity as long as their text spans
overlap.</p>
        <p>The submissions for Task 2 and end-to-end systems are evaluated using
Precision, Recall, and F1-score by comparing the predicted events and gold standard
2 http://chemu.eng.unimelb.edu.au/
events. We consider two events as a match if (1) their trigger words and event
types are the same; and (2) the entities involved in the two events match. Here,
we follow the method in Task 1 to test whether two entities match. This means
that the matching criteria of exact-match and relaxed-match are also applied in
the evaluation of Task 2 and of end-to-end systems. Note that the relaxed-match
will only be applied when matching the spans of two entities; it does not relax
the requirement that the entity type of predicted and ground truth entities must
agree. Since Task 2 provides gold entities but not event triggers with their ground
truth spans, the relaxed-match only re ects the accuracy of spans of predicted
trigger words.</p>
        <p>
          To somewhat accommodate a relaxed form of entity type matching, we also
evaluate submissions in Task 1|NER using a set of high-level labels shown
in the hierarchical structure of entity classes in Fig. 3. The higher-level labels
used are highlighted in grey. In this set of evaluations, given a predicted
entity and a ground-truth entity, we consider that their labels match as long as
their corresponding high-level labels match. For example, suppose we get as
predicted entity \STARTING MATERIAL, [335, 351), boron tribromide" while
the (correct) ground-truth entity instead reads \REAGENT CATALYST, [335,
351), boron tribromide", where each entity is presented in the form of \TYPE,
SPAN, TEXT". In the evaluation framework described earlier this example will
be counted as a mismatch. However, in this additional set of entity type relaxed
evaluations we consider the two entities as a match, since both labels
\STARTING MATERIAL" and \REAGENT CATALYST" specialize their parent label
\COMPOUND".
We released one baseline method for each task as a benchmark method. Speci
cally, the baseline for Task 1 is based on retraining BANNER [
          <xref ref-type="bibr" rid="ref21">21</xref>
          ] on the
training and development data; the baseline for Task 2 is a co-occurrence method; and
the baseline for end-to-end systems is a two-stage algorithm that rst uses
BANNER to identify entities in the input and then uses the co-occurrence method
to extract events.
        </p>
        <p>BANNER. BANNER is a named entity recognition tool for bio-medical
data. In this baseline, we rst use the GENIA Sentence Splitter (GeniaSS) [38]
to split input texts into separate sentences. The resulting sentences are then
fed into BANNER, which predicts the named entities using three steps, namely
tokenization, feature generation, and entity labelling. A simple tokenizer is used
to break sentences into either a contiguous block of letters and/or digits or
a single punctuation mark. BANNER uses a conditional random eld (CRF)
implementation derived from the MALLET toolkit3 for feature generation and
token labelling. The set of machine learning features used consist primarily of
orthographic, morphological and shallow syntax features.</p>
        <p>Co-occurrence Method. This method rst creates a dictionary De for
the observed trigger words and their corresponding types from the training
and development sets. For example, if a word \added" is annotated as a
trigger word with the label of \WORKUP" in the training set, we add an
entry hadded; WORKUPi to De. In the case where the same word has been
observed to appear as both types of \WORKUP" and \REACTION STEP", we
only keep as entry in D its most frequent label. The method also creates an
event dictionary Dr for the observed event types in the training and
development sets. For example, if an event hARG1; E1; E2i is observed where \E1"
corresponds to trigger word \added" of type \WORKUP" and \E2"
corresponds to entity \water" of type \OTHER COMPOUND", we add an entry
hARG1; WORKUP; OTHER COMPOUNDi to Dr.</p>
        <p>To predict events, this method rst identi es all trigger words in the test set
using De. It then extracts two events hARG1; T1; T2i and hARGM; T1; T2i for a
trigger word \E1" and an entity \E2" if (1) they co-occur in the same sentence;
and (2) the relation type hARGx; T1; T2i is included in Dr. Here, \ARGx" can
be \ARG1" or \ARGM", and \T1" and \T2" are the entity types of \E1" and
\E2" respectively.</p>
        <p>BANNER + Co-occurrence Method. The above two baselines are
combined to form a two-stage method for end-to-end systems. This baseline
rst uses BANNER to identify all the entities in Task 1. Then it utilizes the
co-occurrence method to predict events, except that gold standard entities are
replaced with the entities predicted by BANNER in the rst stage.
3 http://mallet.cs.umass.edu/</p>
      </sec>
    </sec>
    <sec id="sec-6">
      <title>Results</title>
      <p>In total, 39 teams registered for the ChEMU shared task, of which 36 teams
registered for Task 1, 31 teams registered for Task 2, and 28 teams registered
for both tasks. The 39 teams are spread across 13 di erent countries, from both
the academic and industry research communities. In this section, we report the
results of all the runs that we received for each task.
Task 1 received considerable interest with the submission of 25 runs from 11
teams. The 11 teams include 1 team from Germany (OntoChem), 3 teams from
India (AUKBC, SSN NLP and JU INDIA), 1 team from Switzerland (BiTeM),
1 team from Portugal (Lasige BioTM), 1 team from Russia (KFU NLP), 1 team
from the United Kingdom (NextMove Software/Minesoft), 2 teams from the
United States of America (Melaxtech and NLP@VCU), and 1 team from
Vietnam (VinAI). We evaluate the performance of all 25 runs, comparing their
predicted entities with the ground-truth entities of the patent snippets in the test
set. We report the performances of all runs under both matching criteria in terms
of three metrics, namely Precision, Recall, and F1-score.</p>
      <p>We report the overall performance of all runs in Table 7. The baseline of
Task 1 achieves 0.8893 in F1-score under exact match. Nine runs outperform the
baseline in terms of F1-score under exact match. The best run was submitted
by team Melaxtech, achieving a high F1-score of 0.9570. There were sixteen
runs with an F1-score greater than 0.90 under relaxed-match. However, under
exact-match, only seven runs surpassed 0.90 in F1-score. This di erence between
exact-match and relaxed-match may be related to the long text spans of chemical
compounds, which is one of the main challenges in NER tasks in the domain of
chemical documents.</p>
      <p>Next, we evaluate the performance of all 25 runs using the high-level labels
in Fig. 3 (highlighted in grey). We report the performances of all runs in terms
of Precision, Recall, and F1-score in Table 8.
We received 10 runs from ve teams. Speci cally, the ve teams include 1 team
from Portugal (Lasige BioTM), 1 team from Turkey (BOUN REX), 1 team
from the United Kingdom (NextMove Software/Minesoft) and 2 teams from
the United States of America (Melaxtech and NLP@VCU). We evaluate all
runs using the metrics Precision, Recall, and F1-score. Again, we utilize the
two matching criteria, namely exact-match and relaxed-match, when comparing
the trigger words in the submitted runs and ground-truth data.</p>
      <p>The overall performance of each run is summarized in Table 9.4 The baseline
(co-occurrence method) scored relatively high in Recall, i.e, 0.8861. This was
4 The run that we received from team Lasige BioTM is not included in the table due
to a technical issue found in this run.</p>
      <p>F</p>
      <p>P</p>
      <p>F</p>
      <p>P</p>
      <p>F
expected, since the co-occurrence method aggressively extracts all possible events
within a sentence. However, the F1-score was low due to its low Precision score.
Here, all runs outperform the baseline in terms of F1-score under exact-match.
Melaxtech ranks rst among all o cial runs in this task, with an F1-score of
0.9536.
We received 10 end-to-end system runs from four teams. The four teams include
The four teams include 1 team from Turkey (BOUN REX), 1 team from the
United Kingdom (NextMove Software/Minesoft) and 2 teams from the United
States of America (Melaxtech and NLP@VCU).</p>
      <p>The overall performance of all runs is summarized in Table 10 in terms of
Precision, Recall, and F1-score under both exact-match and relaxed-match.5
Since gold entities are not provided in this task, the average performance of the
runs in this task are slightly lower than those in Task 2. Note that the Recall
scores of most runs are substantially lower than their Precision scores. This may
reveal that the task of identifying a relation from a chemical patent is harder
5 The run that we received from the Lasige BioTM team is not included in the table
as there was a technical issue in this run. Two runs from Melaxtech,
Melaxtechrun2 and Melaxtech-run3, had very low performance, due to an error in their data
pre-processing step.
than the task of typing an identi ed relation. The rst run from Melaxtech team
ranks best among all runs received for this task.
6</p>
    </sec>
    <sec id="sec-7">
      <title>Overview of Participants' Approaches</title>
      <p>
        We received 8 paper submissions from participating teams, namely BiTeM,
VinAI, BOUN-REX, NextMove/Minesoft, NLP@VCU, AU-KBC, LasigBioTM,
and MelaxTech. In this section, we present an overview of the approaches
proposed by these teams. We start by introducing the approach of each team rst.
Then we discuss the di erences between these approaches.
To tackle the complexities of chemical patent narratives, the BiTeM team
explored the power of ensemble of deep neural language models based on
transformer architectures to extract information in chemical patents [
        <xref ref-type="bibr" rid="ref10">10</xref>
        ]. Using a
majority vote strategy [
        <xref ref-type="bibr" rid="ref9">9</xref>
        ], their approach combined end-to-end architectures,
including Bidirectional Encoder Representations from Transformers (BERT)
models (including both base/large and cased/uncased) [
        <xref ref-type="bibr" rid="ref12">12</xref>
        ], the ChemBERTa model6,
and a model based on Convolutional Neural Network (CNN) [
        <xref ref-type="bibr" rid="ref22">22</xref>
        ] fed with
contextualized embedding vectors provided by BERT model. To learn to classify
chemical entities in patent passages, the language models were ne-tuned to
6 https://github.com/seyonechithrananda/bert-loves-chemistry
categorize tokens using training examples of the ChEMU NER task. The best
model proposed by BiTeM { an ensemble of BERT-base cased and uncased, and
a CNN { achieved 92.3% of exact F1-score and 96.24% of relaxed F1-score in
the test phase, outperforming the exact F1-score of the best individual model
(BERT-base cased) by 1.3% and the challenge's baseline by 3.4%. The results
of BiTeM team show that ensemble of contextualized language models could be
used to e ectively detect chemical entities in patent narratives.
Following [48], the VinAI system employed the well-known BiLSTM-CNN-CRF
model [
        <xref ref-type="bibr" rid="ref26">26</xref>
        ] with additional contextualized word embeddings. In particular, given
an input sequence of words, VinAI represented each word token by
concatenating its corresponding pre-trained word embedding, CNN-based character-level
word embedding [
        <xref ref-type="bibr" rid="ref26">26</xref>
        ] and contextualized word embedding. Here, VinAI used
the pre-trained word embeddings released by [48], which are trained on a
corpus of 84K chemical patents using the Word2Vec skip-gram model [
        <xref ref-type="bibr" rid="ref28">28</xref>
        ]. Also,
VinAI employed the contextualized word embeddings generated by a pre-trained
ELMo language model [
        <xref ref-type="bibr" rid="ref35">35</xref>
        ] which is trained using the same corpus of 84K
chemical patents [48].7 Then the concatenated word representations are fed into a
BiLSTM encoder to extract latent feature vectors for input words. Each latent
feature vector is then linearly transformed before being fed into a linear-chain
CRF layer for NER label prediction [
        <xref ref-type="bibr" rid="ref19">19</xref>
        ]. VinAI achieved very high performance,
o cially ranking second with regards to both exact- and relaxed-match F1-scores
at 94.33% and 96.84%, respectively. In a post-evaluation phase, xing a
mapping bug which converted the column-based format into the brat stando
format helped VinAI to obtain higher results: an exact-match F1-score at 95.21%
and especially a relaxed-match F1-score at 97.26%, thus achieving the highest
relaxed-match F1-score compared with all other participating systems.
The BOUN-REX system addressed the event extraction task with two steps: the
detection of trigger word and its trigger type, and identi cation of the event type.
A pre-trained transformer-based model, BioBERT, was used for the detection
of trigger words (and the exact types of the detected trigger words) whereas the
event type is determined using a rule-based method. Speci cally, to pre-process
the dataset, the documents were rst split into sentences via GENIA Sentence
Splitter. After constructing sentence-entity pairs for each entity, events and
trigger words were predicted from the given sentence-entity pairs. Start and end
markers were also introduced for each entity to indicate position of an entity in
a sentence. A pre-trained transformer-based architecture was constructed as a
base model to extract a xed-length relation representation and token
representations from an input sentence with entity markers. The xed-length relation
7 https://github.com/zenanz/ChemPatentEmbeddings
representation was utilized to detect the type of the trigger word. In addition, a
trigger word span model was also constructed to predict the probabilities of start
and end markers of trigger words with the token representations. The system
trained using an AdamW optimizer, achieving the best performance of an F1
score at 0.7234 using exact match.
      </p>
      <sec id="sec-7-1">
        <title>NextMove Software/Minesoft</title>
        <p>
          Lowe and May eld [
          <xref ref-type="bibr" rid="ref24">24</xref>
          ] used an approach utilizing grammars both for
recognizing entities and for determining the relationships between entities. The toolkit
LeadMine was rst used to recognized chemicals and physical quantities, which
is achieved by employing e cient matching against extensive grammars that
describe these entity types. These entities were used as the highest priority tagger
in an enhanced version of ChemicalTagger, with ChemicalTagger's default rule
based tokenization being adjusted such that each LeadMine entity was a single
token. Remaining tokens were assigned tags from pattern matches, or failing
that assigned a part of speech tag. The pattern matches notably are how the
reaction action trigger words are detected. An Antlr grammar arranges the tagged
tokens into a parse tree which groups at various levels of granularity, e.g. all
tokens for a particular reaction action will be grouped. The parse tree is used
to determine which chemicals are solvents or involved in workup actions. The
chemical structures (determined from the names), is used to assign chemical role
information, both by inspection of the individual compounds and through whole
reaction analysis techniques like NameRxn and atom-atom mapping. From
analysis of the whole reaction the stoichiometry of the reaction is determined, hence
distinguishing catalysts from starting materials.
6.5
        </p>
        <p>
          VLP@VCU
VLP@VCU team participated in two tasks: Task 1|NER and Task 2{EE. For
Task 1, the VLP@VCU team identi ed the named entities using BiLSTM units
with a Conditational Random Graph (CRF) output layer. The inputs to this
model are pre-trained word embeddings [
          <xref ref-type="bibr" rid="ref32">32</xref>
          ] in combination with character
embeddings. These embeddings are concatenated and then passed through the
BiLSTM network. The VLP@VCU system achieved an overall performance with a
precision, recall and F1-score at 0.87, 0.86, and 0.87 in terms of exact match,
and a precision, recall and F1-score of 0.95, 0.99, and 0.97 in terms of relaxed
match.
        </p>
        <p>
          For Task 2|EE, the VLP@VCU team explored two methods to identify
the chemical arguments between the trigger words and the entities. First, a
rule-based method was explored, which uses a breadth- rst search to nd the
closest occurrence of the trigger word on either side of the entity. Second, a
CNNbased model was explored. This model performs a binary classi cation to identify
whether there is a relation or not for each Trigger word-Entity pair. The
sentence containing the Trigger word-Entity pair is rst extracted and then divided
into ve segments, where each segment is represented by a k N matrix. Here,
k represents the latent dimensionality of the pre-trained word embeddings [
          <xref ref-type="bibr" rid="ref32">32</xref>
          ]
and N is the number of words in the segment. A separate convolution unit is
constructed for each segment, the outputs of which are then attened,
concatenated, and fed into the fully connected feedforward layer. Finally, the output of
the fully connected feedforward layer is fed into a softmax layer, which performs
the nal classi cation. This CNN-based method obtained higher performance
with a precision, recall and F1-score of 0.80, 0.54 and 0.65, respectively.
The AU-KBC team submitted two systems that were developed with two
different Machine Learning (ML) techniques: CRFs and Arti cial Neural Networks
(ANNs). A two-stage pre-processing was done on the training and development
data sets: (1) a formatting stage that consists of three steps, i.e., sentence
splitting, tokenization, and conversion of data format to column; and (2) a data
annotation stage, where the data is annotated for syntactic information, including
Part-of-Speech (PoS) and Phrase Chunk information (noun phrase, verb phrase).
To extract the PoS and chunk information, an open source tool, fnTBL [
          <xref ref-type="bibr" rid="ref31">31</xref>
          ], is
used. Three types of features were used for training: (a) word-level features, (b)
grammatical features, and (c) functional terms features. Speci cally, word-level
features include orthographical features (e.g., capitalization, Greek words, and
combination of digits, symbols, and words) and morphological features (e.g.,
common pre xes and su xes of chemical entities). Grammatical features
include word, Part-of-Speech (PoS), chunks and combination of PoS and chunks.
Functional terms were used to help identify the biological named entities and
assign them with correct labels. After extraction of these linguistic features, two
models based on CRF and ANN are built to address Task 1. Note that the two
models only utilized the training data provided in the task and did not rely
on any external resources or leverage pre-trained language models. Speci cally,
the CRF++ tool [
          <xref ref-type="bibr" rid="ref2">2</xref>
          ] was used for developing the CRF model and the ANN
model was implemented using the scikit python package. The ANN model is a
Multi-Layer Perceptron (MLP), where ReLU activation function was used. The
stochastic gradient Adam optimizer was used for optimizing weights of the ANN
model. We obtained an F1-score of 0.6640 using CRFs and F1-score of 0.3764
using ANN.
To address Task 1, the LasigBioTM team ne-tuned the BioBERT NER model in
the train set plus half of the development set (Note that the data was converted
to the IOB2 format), and applied the ne-tuned model into the second half of
the development set for recognizing and locating named entities. The team also
developed a module to handle the BioBERT output and to generate the
annotation les in the BRAT format and then submitted them to the competition page
for evaluation, obtaining an F1-score of 0.9524 using the exact matching criterion
and a F1-score of 0.9904 using the relaxed matching criterion on development
data. In the testing phase, the team ne-tuned the model again, but using all
the documents belonging to the train and the development sets. For Task 2, the
team considered the BioBERT NER model jointly with the BioBERT RE model.
They followed a similar approach as for Task 1 to detect the trigger words. To
further extract the relations between triggers and entities, the team performed
sentence segmentation of the train and the development sets and, if a trigger
word and an entity were present in a given sentence, a relation was assumed
to exist between them if it was referred in the respective annotation le. The
BioBERT RE model was also ne-tuned using the sentences of the train and the
development sets.
The MelaxTech system is a hybrid combination of deep learning models and
pattern-based rules for this task. For deep learning, a language model of patents
with chemical reactions was rst built. Speci cally, the BioBERT [
          <xref ref-type="bibr" rid="ref23">23</xref>
          ], a
pretrained biomedical language model (a bidirectional encoder representation for
biomedical text), was used as the basis for training a language model of patents.
Based on BERT [
          <xref ref-type="bibr" rid="ref12">12</xref>
          ], a language model pre-trained on large scale open text,
BioBERT was further re ned using the biomedical literature in PubMed and
PMC. For this study, BioBERT was retrained on patent data to generate a new
language model of Patent BioBERT. For the NER subtask, Patent BioBERT
was ne-tuned using the Bi-LSTM-CRF (Bi-directional Long-Short-Term-Memory
Conditional-Random-Field) algorithm [
          <xref ref-type="bibr" rid="ref26">26</xref>
          ]. Next, several rules based on observed
patterns in the training data were used in a post-processing step. For example,
rules were de ned to di erentiate STARTING MATERIAL and OTHER
COMPOUND based on the relative positions and total number of EXAMPLE LABEL
occurrences. For the event extraction subtask, the event triggers were rst
identi ed as named entities together with other semantic roles in chemical reaction,
using the same approach as in the NER subtask. Next, a binary classi er was
built by ne-tuning Patent BioBERT to recognize relations between event
triggers and semantic roles in the same sentence. Some event triggers and their
linked semantic roles were present in di erent sentences, or di erent clauses
in long complex sentences. Their relations were not identi ed using the deep
learning-based model. Therefore, post-processing rules were designed based on
patterns observed in the training data, and applied to recover some of these
false negative relations. The proposed approaches demonstrated promising
results, which achieved top ranks in both subtasks, with the best F1-score of 0.957
for entity recognition and the best F1-score of 0.9536 for event extraction.
7
        </p>
      </sec>
    </sec>
    <sec id="sec-8">
      <title>Discussion</title>
      <p>Di erent approaches were explored by the participating teams. In Table 11, we
summarize the key strategies in terms of three aspects: tokenization method,
token representations, and core model architecture.</p>
      <p>For teams who participated in Tasks 2 and 3, a common two-step strategy was
adopted for relation extraction: (1) identify trigger words; and (2) extract the
relation between identi ed trigger words and entities. The rst step is essentially
an NER task, and the second step is a relation extraction task. As such, NER
models were used by all these teams for Tasks 2 and 3 as well as by the teams
participating in Task 1. Therefore, in what follows, we rst discuss and compare
the approaches of all teams without considering the target tasks, subsequently
considering relation extraction approaches.</p>
      <p>Chemistry domain-speci c
Representation</p>
      <p>Embeddings</p>
      <p>Character-level
Pre-trained</p>
      <p>Chemistry domain-speci c
Features</p>
      <p>PoS</p>
      <p>Phrase
Model Architecture</p>
      <p>
        Transformer
Bi-LSTM
CNN
MLP
CRF
FSM
Rule based
Tokenization is an important data pre-processing step that splits input texts
into words/subwords, i.e., tokens. We identify three general types of
tokenization methods used by participants: (1) rule based tokenization; (2) dictionary
based tokenization; and (3) subword based tokenization. Speci cally, rule based
tokenization applies pre-de ned rules to split texts into tokens. The rules applied
can be as simple as \white-space tokenization", but can also be a complex
mixture of a set of carefully designed rules (e.g., based on language-speci c grammar
rules and common pre xes). Dictionary based tokenization requires the
construction of a vocabulary and the text splitting is performed by matching the input
text with the existing tokens in the constructed vocabulary. Subword
tokenization allows a token to be a sub-string of a word, i.e., subword units. It relies
on the principle that most common words should be left as is, but rare words
should be decomposed in meaningful subword units. Popular subword
tokenization methods include WordPiece [39] and Byte Pair Encoding (BPE) [
        <xref ref-type="bibr" rid="ref36">36</xref>
        ]. For
each participating team, we consider whether their approach belong to one or
multiple of the three categories, and summarize our ndings in Table 11. Finally,
we also indicate whether their tokenization methods consider domain-speci c
knowledge in Table 11.
      </p>
      <p>
        Four teams utilized tokenization methods that are purely rule-based.
Specifically, VinAI used Oscar4 tokenizer [
        <xref ref-type="bibr" rid="ref15">15</xref>
        ]. This tokenizer is particularly designed
for chemical texts, and is made up by a set of pattern matching rules (e.g.,
pre x matching) that are designed based on domain knowledge from
chemical experts. NLP@VCU used Spacy tokenizer [
        <xref ref-type="bibr" rid="ref4">4</xref>
        ], which consists of a collection
of complex normalization and segmentation logic and has been proven to work
well with general English corpus. NextMove/Minesoft used a combination of
Oscar4 and LeadMine [
        <xref ref-type="bibr" rid="ref25">25</xref>
        ] tokenzier. LeadMine was rst run on untokenized text
to identify entities using auxiliary grammars or dictionaries. Oscar4 was then
used for general tokenization but is adjusted so that each entity recognized by
LeadMine corresponds to exactly one token. Four teams, BiTeM, BOUN-REX,
LasigBioTM, and MelaxTech, chose to leverage the pre-trained model BERT
(or variants of BERT) to address our tasks, and thus, the four teams used the
subword-based tokenizer, WordPiece, that is built-in within BERT. BOUN-REX,
LasigBioTM and MelaxTech used BioBERT model which is language model
pretrained on biomedical texts. Since this model is a continual training based on
the original BERT model, the vocabulary used in BioBERT does not di er from
BERT, i.e., domain-speci c tokenization is not used. However, since MelaxTech
performed a pre-tokenization using a toolkit CLAMP [41], we consider their
approach as domain-speci c, since CLAMP is tailored for clinical texts. BiTeM
used the model ChemBERTa that is pre-trained on ZINC corpus. It is unclear
yet whether the tokenization is domain-speci c due to the lack of documentation
of ChemBERTa. Finally, since WordPiece needs an extra pre-tokenization step,
we consider it as a hybrid of rule-based and subword-based method.
When transforming tokens into machine-readable representations, two types of
methods are used: (1) feature extraction that represents tokens with their
linguistic characteristics such as word-level features (e.g., morphological features)
and grammatical features (e.g., PoS tags); and (2) embedding methods in which
token representations are randomly initialized as numerical vectors (or initialized
from pre-trained embeddings) and then learned (or ne-tuned) from provided
training data. Two teams, NextMove/Minesoft and AU-KBC adopted the rst
strategy and the other teams adopted the second strategy. Among the teams that
used embeddings to represent tokens, two teams, VinAI and VLP@VCU further
added character-level embeddings to their systems. All of these six teams used
pre-trained embeddings, and ve teams used embeddings that are pre-trained
for related domains: VinAI and NLP@VCU used the embeddings that are
pretrained on chemical patents [48], BOUN-REX and LasigBioTM used the
embeddings from BioBERT model that are pre-trained on PubMed corpus. MexlaxTech
also used embeddings from BioBERT, but they further tuned the embeddings
using the patent documents released in the test phase.
Various architectures were employed by participating teams. Four teams, BiTeM,
BOUN-REX, LasigBioTM, and MelaxTech developed their systems based on
transformers [43]. BiTem submitted an additional run using an ensemble of
a Transformer-based model and a CNN-based model. They also had a third
run that is built based on CRF. The other two teams MelaxTech and
BOUNREX added rule-based techniques into their systems. MelaxTech added several
pattern-matching rules in their post-processing step. BOUN-REX focused on
Task 2 and their system used rule based methods to determine the event type
of each detected event. Two teams, VinAI and NLP@VCU, used the
architecture of BiLSTM-CNN-CRF for Task 1. NLP@VCU also participated in Task
2 and they proposed two systems based on rules, and CNN architecture,
respectively. NextMove/Minesoft utilized Chemical Tagger [
        <xref ref-type="bibr" rid="ref13">13</xref>
        ], a model based
on Finite State Machine (FSM), and a set of comprehensive rules are applied
to generate predictions. AU-KBC proposed two systems for Task 1, based on
multi-layer perceptron and CRF, respectively.
      </p>
      <sec id="sec-8-1">
        <title>Approaches to relation extraction</title>
        <p>Four of the above teams participated in Task 2 or Task 3. As mentioned before,
these teams utilized their NER models for trigger word detection. Thus, here, we
only discuss their approaches for relation extraction assuming that the trigger
words and entities are known.</p>
        <p>NextMove/Minesoft again made use of ChemicalTagger for event extraction.
ChemicalTagger is able to recognize WORKUP and REACTION STEP words,
thus, assignment of relationships were achieved by associating all entities in a
ChemicalTagger action phrase with the trigger word responsible for the action
phrase. A set of post-processing rules were also applied to enhance the accuracy
of ChemicalTagger.</p>
        <p>LasigBioTM, NLP@VCU, and MelaxTech formulated the task of relation
extraction as a binary classi cation problem. That is, given each candidate pair
of trigger word and named entity that co-locate within an input sentence, the
goal of the task is to determine whether the candidate pair of entities are related
or not.</p>
        <p>LasigBioTM developed a BioBERT-based model to accomplish this classi
cation. The input of BioBERT is the sentence containing the candidate pair but
the trigger word and named entity of the candidate pair were replaced with the
tags \@TRIGGER$" and \@LABEL$", respectively. The output of BioBERT is
modi ed as a binary classi cation layer which aims to predict the existence of
relation for the candidate pair.</p>
        <p>NLP@VCU proposed two systems for relation extraction. Their rst system
is a rule-based system. Given a named entity, a relation is extracted between
the named entity and its nearest trigger word. Their second system is developed
based on CNNs. They split the sentence containing the candidate pair into ve
segments: the sequence of tokens before/between/after the candidate pair, the
trigger word, and the named entity of the candidate pair. Separate convolutional
units were used to learn the latent representations of the ve segments, and a
nal output layer was used to determine if the candidate pair is related or not.</p>
        <p>MelaxTech continued the use of the BioBERT model re-trained on the patent
texts released during the test phase. Similar to LasigBioTM, the input to their
model is the sentence containing the candidate pair but only the candidate
named entity is generalized by its semantic type in the sentences. Furthermore,
rules were also applied in the post-processing step to recover false negative
relations with a long distance, including relations across clauses and across sentences.
7.5</p>
      </sec>
      <sec id="sec-8-2">
        <title>Summary of observations</title>
        <p>The various approaches adopted by teams and the resulting performances have
provided us with valuable experiences in how to address the tasks and what
choices of methods are more suitable for our tasks.</p>
        <p>Tokenization. In general, domain-speci c tokenization tools perform
better than tokenization methods that are for general English corpus. This is as
expected since the vocabulary of chemical patents contains a large number of
domain-speci c terminology, and a machine can better understand and learn
the characteristics of input texts if the texts are split into meaningful tokens.
Another observation is that subword-based tokenization may contribute to
overall accuracy. Chemical names are usually long, which make subword-based
tokenization a suitable method for breaking down long chemical names. But further
investigation is needed to support this claim.</p>
        <p>
          Representation. Pre-trained embeddings are shown to be e ective in
enhancing system performances. Speci cally, the Melaxtech and Lasige BioTM
systems are based on BioBERT model [
          <xref ref-type="bibr" rid="ref23">23</xref>
          ] and ranked the rst and third place
in Task 1. The VinAI system leveraged embeddings pre-trained on chemical
patents [48] and ranked second place. Character-level embeddings are also
bene cial, shown by the ablation study in [
          <xref ref-type="bibr" rid="ref11">11</xref>
          ] and [
          <xref ref-type="bibr" rid="ref27">27</xref>
          ].
        </p>
        <p>
          Model Architecture. The most popular choice of model is BERT [
          <xref ref-type="bibr" rid="ref12">12</xref>
          ],
which is based on Transformer [43]. The model has demonstrated its e ectiveness
in sequence learning again. The Melaxtech system adopted this architecture and
ranked rst place in all three tasks. However, it is also worthwhile to note that
the architecture of BiLSTM-CNN-CRF is still very competitive with BERT. The
VinAI system ranked the rst place in F1-score when relaxed-match is used.
8
        </p>
      </sec>
    </sec>
    <sec id="sec-9">
      <title>Conclusions</title>
      <p>This paper presents a general overview of the activities and outcomes of the
ChEMU 2020 evaluation lab. The ChEMU lab targets two important
information extraction tasks applied to chemical patents: (1) named entity recognition,
which aims to identify chemical compounds and their speci c roles in chemical
reactions; and (2) event extraction, which aims to identify the single event steps
that form a chemical reaction.</p>
      <p>We received registrations from 39 teams and 46 runs from 11 teams across
all tasks and tracks, and 8 teams have contributed detailed system descriptions
for their methods. The evaluation results show that many e ective solutions
have been proposed, with systems achieving excellent performance on each task,
up to nearly 0.98 macro-averaged F1-score on the NER task (and up to 0.99
F1-score on a relaxed match), 0.95 F1-score on the isolated relation extraction
task, and around 0.92 F1-score for the end-to-end systems. These results strongly
outperformed baselines.</p>
    </sec>
    <sec id="sec-10">
      <title>Acknowledgements</title>
      <p>We are grateful for the detailed excerption and annotation work of the domain
experts that support Reaxys, and the support of Ivan Krstic, Director of
Chemistry Solutions at Elsevier. Funding for the ChEMU project is provided by an
Australian Research Council Linkage Project, project number LP160101469, and
Elsevier.
37. Ruas, P., Lamurias, A., Couto, F.M.: LasigeBioTM team at CLEF2020 ChEMU
evaluation lab: Named Entity Recognition and Event extraction from chemical
reactions described in patents using BioBERT NER and RE. In: Working Notes
of CLEF 2020|Conference and Labs of the Evaluation Forum (2020)
38. S tre, R., Yoshida, K., Yakushiji, A., Miyao, Y., Matsubayashi, Y., Ohta, T.:
AKANE system: protein-protein interaction pairs in BioCreAtIvE2 challenge,
PPI-IPS subtask. In: Proceedings of the second BioCreative challenge workshop.
vol. 209, p. 212. Madrid (2007)
39. Schuster, M., Nakajima, K.: Japanese and korean voice search. In: 2012 IEEE
International Conference on Acoustics, Speech and Signal Processing (ICASSP).
pp. 5149{5152. IEEE (2012)
40. Senger, S., Bartek, L., Papadatos, G., Gaulton, A.: Managing expectations:
assessment of chemistry databases generated by automated extraction of chemical
structures from patents. Journal of Cheminformatics 7(1), 1{12 (2015)
41. Soysal, E., Wang, J., Jiang, M., Wu, Y., Pakhomov, S., Liu, H., Xu, H.: CLAMP{
a toolkit for e ciently building customized clinical natural language processing
pipelines. Journal of the American Medical Informatics Association 25(3), 331{
336 (2018)
42. Stenetorp, P., Pyysalo, S., Topic, G., Ohta, T., Ananiadou, S., Tsujii, J.: BRAT:
a web-based tool for NLP-assisted text annotation. In: Proceedings of the
Demonstrations at the 13th Conference of the European Chapter of the Association for
Computational Linguistics. pp. 102{107 (2012)
43. Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser,
L., Polosukhin, I.: Attention is all you need. In: Advances in neural information
processing systems. pp. 5998{6008 (2017)
44. Verspoor, C.M.: Contextually-dependent lexical semantics (1997)
45. Verspoor, K., Nguyen, D.Q., Akhondi, S.A., Druckenbrodt, C., Thorne, C., Hoessel,
R., He, J., Zhai, Z..: ChEMU dataset for information extraction from chemical
patents. https://doi.org/10.17632/wy6745bjfj.1
46. Wang, J., Ren, Y., Zhang, Z., Zhang, Y.: Melaxtech: A report for CLEF 2020 {
ChEMU Task of Chemical Reaction Extraction from Patent. In: Working Notes of
CLEF 2020|Conference and Labs of the Evaluation Forum (2020)
47. Yoshikawa, H., Nguyen, D.Q., Zhai, Z., Druckenbrodt, C., Thorne, C., Akhondi,
S.A., Baldwin, T., Verspoor, K.: Detecting Chemical Reactions in Patents. In:
Proceedings of the 17th Annual Workshop of the Australasian Language Technology
Association. pp. 100{110 (2019)
48. Zhai, Z., Nguyen, D.Q., Akhondi, S., Thorne, C., Druckenbrodt, C., Cohn, T.,
Gregory, M., Verspoor, K.: Improving Chemical Named Entity Recognition in Patents
with Contextualized Word Embeddings. In: Proceedings of the 18th BioNLP
Workshop and Shared Task. pp. 328{338. Association for Computational Linguistics
(2019)</p>
    </sec>
  </body>
  <back>
    <ref-list>
      <ref id="ref1">
        <mixed-citation>
          1.
          <article-title>BRATEval evaluation tool</article-title>
          . https://bitbucket.org/nicta_biomed/brateval/ src/master/
        </mixed-citation>
      </ref>
      <ref id="ref2">
        <mixed-citation>
          2. CRF++ Toolkit. https://taku910.github.io/crfpp/, accessed:
          <fpage>2020</fpage>
          -06-23
        </mixed-citation>
      </ref>
      <ref id="ref3">
        <mixed-citation>
          3.
          <string-name>
            <given-names>International</given-names>
            <surname>Patent</surname>
          </string-name>
          <article-title>Classi cation</article-title>
          . https://www.wipo.int/classifications/ ipc/en/
        </mixed-citation>
      </ref>
      <ref id="ref4">
        <mixed-citation>4. Spacy tokenizer. https://spacy.io/api/tokenizer</mixed-citation>
      </ref>
      <ref id="ref5">
        <mixed-citation>
          5.
          <string-name>
            <surname>Akhondi</surname>
            ,
            <given-names>S.A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Klenner</surname>
            ,
            <given-names>A.G.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Tyrchan</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Manchala</surname>
            ,
            <given-names>A.K.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Boppana</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Lowe</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Zimmermann</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Jagarlapudi</surname>
            ,
            <given-names>S.A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Sayle</surname>
            ,
            <given-names>R.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Kors</surname>
            ,
            <given-names>J.A.</given-names>
          </string-name>
          , et al.:
          <article-title>Annotated chemical patent corpus: a gold standard for text mining</article-title>
          .
          <source>PLoS One</source>
          <volume>9</volume>
          (
          <issue>9</issue>
          ),
          <year>e107477</year>
          (
          <year>2014</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref6">
        <mixed-citation>
          6.
          <string-name>
            <surname>Akhondi</surname>
            ,
            <given-names>S.A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Rey</surname>
            ,
            <given-names>H.</given-names>
          </string-name>
          , Schworer,
          <string-name>
            <given-names>M.</given-names>
            ,
            <surname>Maier</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            ,
            <surname>Toomey</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            ,
            <surname>Nau</surname>
          </string-name>
          ,
          <string-name>
            <given-names>H.</given-names>
            ,
            <surname>Ilchmann</surname>
          </string-name>
          ,
          <string-name>
            <given-names>G.</given-names>
            ,
            <surname>Sheehan</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            ,
            <surname>Irmer</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            ,
            <surname>Bobach</surname>
          </string-name>
          ,
          <string-name>
            <surname>C.</surname>
          </string-name>
          , et al.:
          <article-title>Automatic identi cation of relevant chemical compounds from patents</article-title>
          .
          <source>Database</source>
          <year>2019</year>
          (
          <year>2019</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref7">
        <mixed-citation>
          7.
          <string-name>
            <surname>Bregonje</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          :
          <article-title>Patents: A unique source for scienti c technical information in chemistry related industry?</article-title>
          <source>World Patent Information</source>
          <volume>27</volume>
          (
          <issue>4</issue>
          ),
          <volume>309</volume>
          {
          <fpage>315</fpage>
          (
          <year>2005</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref8">
        <mixed-citation>
          8.
          <string-name>
            <surname>Carletta</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          :
          <article-title>Assessing agreement on classi cation tasks: The kappa statistic</article-title>
          .
          <source>Computational Linguistics</source>
          <volume>22</volume>
          (
          <issue>2</issue>
          ),
          <volume>249</volume>
          {
          <fpage>254</fpage>
          (
          <year>1996</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref9">
        <mixed-citation>
          9.
          <string-name>
            <surname>Copara</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Knafou</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Naderi</surname>
            ,
            <given-names>N.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Moro</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Ruch</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Teodoro</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          :
          <article-title>Contextualized french language models for biomedical named entity recognition</article-title>
          . In: Actes de la 6e conference
          <article-title>conjointe Journees d'Etudes sur la Parole (JEP, 31e edition), Traitement Automatique des Langues Naturelles (TALN, 27e edition), Rencontre des Etudiants Chercheurs en Informatique pour le Traitement Automatique des Langues (RECITAL, 22e edition)</article-title>
          . Atelier DE Fouille de Textes. pp.
          <volume>36</volume>
          {
          <fpage>48</fpage>
          .
          <string-name>
            <surname>ATALA</surname>
          </string-name>
          (
          <year>2020</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref10">
        <mixed-citation>
          10.
          <string-name>
            <surname>Copara</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Naderi</surname>
            ,
            <given-names>N.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Knafou</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Ruch</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Teodoro</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          :
          <article-title>Named entity recognition in chemical patents using ensemble of contextual language models</article-title>
          .
          <source>In: Working Notes of CLEF</source>
          <year>2020</year>
          |
          <article-title>Conference and Labs of the Evaluation Forum (</article-title>
          <year>2020</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref11">
        <mixed-citation>
          11.
          <string-name>
            <surname>Dao</surname>
            ,
            <given-names>M.H.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Nguyen</surname>
            ,
            <given-names>D.Q.</given-names>
          </string-name>
          : VinAI at ChEMU 2020:
          <article-title>An accurate system for named entity recognition in chemical reactions from patents</article-title>
          .
          <source>In: Working Notes of CLEF</source>
          <year>2020</year>
          |
          <article-title>Conference and Labs of the Evaluation Forum (</article-title>
          <year>2020</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref12">
        <mixed-citation>
          12.
          <string-name>
            <surname>Devlin</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Chang</surname>
            ,
            <given-names>M.W.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Lee</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Toutanova</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          :
          <article-title>Bert: Pre-training of deep bidirectional transformers for language understanding</article-title>
          .
          <source>In: Proceedings of NAACL-HLT</source>
          . pp.
          <volume>4171</volume>
          {
          <issue>4186</issue>
          (
          <year>2019</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref13">
        <mixed-citation>
          13.
          <string-name>
            <surname>Hawizy</surname>
            ,
            <given-names>L.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Jessop</surname>
            ,
            <given-names>D.M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Adams</surname>
            ,
            <given-names>N.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Murray-Rust</surname>
            ,
            <given-names>P.:</given-names>
          </string-name>
          <article-title>ChemicalTagger: A tool for semantic text-mining in chemistry</article-title>
          .
          <source>Journal of cheminformatics 3(1)</source>
          ,
          <volume>17</volume>
          (
          <year>2011</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref14">
        <mixed-citation>
          14.
          <string-name>
            <surname>He</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Nguyen</surname>
            ,
            <given-names>D.Q.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Akhondi</surname>
            ,
            <given-names>S.A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Druckenbrodt</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Thorne</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Hoessel</surname>
            ,
            <given-names>R.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Afzal</surname>
            ,
            <given-names>Z.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Zhai</surname>
            ,
            <given-names>Z.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Fang</surname>
            ,
            <given-names>B.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Yoshikawa</surname>
            ,
            <given-names>H.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Albahem</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Cavedon</surname>
            ,
            <given-names>L.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Cohn</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Baldwin</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Verspoor</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          :
          <article-title>Overview of chemu 2020: Named entity recognition and event extraction of chemical reactions from patents</article-title>
          .
          <source>In: Experimental IR Meets Multilinguality, Multimodality, and Interaction. Proceedings of the Eleventh International Conference of the CLEF Association (CLEF</source>
          <year>2020</year>
          ), vol.
          <volume>12260</volume>
          . Lecture Notes in Computer Science (
          <year>2020</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref15">
        <mixed-citation>
          15.
          <string-name>
            <surname>Jessop</surname>
            ,
            <given-names>D.M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Adams</surname>
            ,
            <given-names>S.E.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Willighagen</surname>
            ,
            <given-names>E.L.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Hawizy</surname>
            ,
            <given-names>L.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Murray-Rust</surname>
            ,
            <given-names>P.:</given-names>
          </string-name>
          <article-title>OSCAR4: a exible architecture for chemical text-mining</article-title>
          .
          <source>Journal of cheminformatics 3(1)</source>
          ,
          <volume>1</volume>
          {
          <fpage>12</fpage>
          (
          <year>2011</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref16">
        <mixed-citation>
          16.
          <string-name>
            <surname>Jurafsky</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Martin</surname>
            ,
            <given-names>J.H.</given-names>
          </string-name>
          :
          <article-title>Speech &amp; Language Processing, 3rd edition, chap</article-title>
          .
          <source>Semantic Role Labeling and Argument Structure. Pearson Education India</source>
          (
          <year>2009</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref17">
        <mixed-citation>
          17.
          <string-name>
            <surname>Kim</surname>
            ,
            <given-names>J.D.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Ohta</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Pyysalo</surname>
            ,
            <given-names>S.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Kano</surname>
            ,
            <given-names>Y.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Tsujii</surname>
          </string-name>
          , J.:
          <article-title>Overview of BioNLP'09 shared task on event extraction</article-title>
          .
          <source>In: Proceedings of the BioNLP 2009 workshop companion volume for shared task</source>
          . pp.
          <volume>1</volume>
          {
          <issue>9</issue>
          (
          <year>2009</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref18">
        <mixed-citation>
          18. Koksal,
          <string-name>
            <given-names>A.</given-names>
            ,
            <surname>Hilal</surname>
          </string-name>
          ,
          <string-name>
            <surname>D.</surname>
          </string-name>
          ,
          <string-name>
            <surname>O</surname>
          </string-name>
          <article-title>zk r ml Elif, Ozgur Arzucan: BOUN-REX at CLEF-2020 ChEMU Task 2: Evaluating Pretrained Transformers for Event Extraction</article-title>
          . In: Working Notes of CLEF 2020|
          <article-title>Conference and Labs of the Evaluation Forum (</article-title>
          <year>2020</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref19">
        <mixed-citation>
          19.
          <string-name>
            <surname>La</surname>
            <given-names>erty</given-names>
          </string-name>
          , J.D.,
          <string-name>
            <surname>McCallum</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Pereira</surname>
            ,
            <given-names>F.C.N.</given-names>
          </string-name>
          :
          <article-title>Conditional Random Fields: Probabilistic Models for Segmenting and Labeling Sequence Data</article-title>
          .
          <source>In: Proceedings of the 18th International Conference on Machine Learning</source>
          . pp.
          <volume>282</volume>
          {
          <issue>289</issue>
          (
          <year>2001</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref20">
        <mixed-citation>
          20.
          <string-name>
            <surname>Lawson</surname>
            ,
            <given-names>A.J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Roller</surname>
            ,
            <given-names>S.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Grotz</surname>
            ,
            <given-names>H.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Wisniewski</surname>
            ,
            <given-names>J.L.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Goebels</surname>
            ,
            <given-names>L.</given-names>
          </string-name>
          :
          <article-title>Method and software for extracting chemical data. German patent no</article-title>
          .
          <source>DE102005020083A1</source>
          (
          <year>2011</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref21">
        <mixed-citation>
          21.
          <string-name>
            <surname>Leaman</surname>
            ,
            <given-names>R.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Gonzalez</surname>
          </string-name>
          , G.:
          <article-title>BANNER: an executable survey of advances in biomedical named entity recognition</article-title>
          .
          <source>In: Paci c Symposium on Biocomputing</source>
          <year>2008</year>
          , pp.
          <volume>652</volume>
          {
          <fpage>663</fpage>
          . World Scienti c (
          <year>2008</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref22">
        <mixed-citation>
          22.
          <string-name>
            <surname>Lecun</surname>
            ,
            <given-names>Y.</given-names>
          </string-name>
          :
          <article-title>Generalization and network design strategies</article-title>
          .
          <source>Technical Report CRGTR-89-4</source>
          , University of Toronto (
          <year>June 1989</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref23">
        <mixed-citation>
          23.
          <string-name>
            <surname>Lee</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Yoon</surname>
            ,
            <given-names>W.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Kim</surname>
            ,
            <given-names>S.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Kim</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Kim</surname>
            ,
            <given-names>S.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>So</surname>
            ,
            <given-names>C.H.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Kang</surname>
          </string-name>
          , J.:
          <article-title>Biobert: a pre-trained biomedical language representation model for biomedical text mining</article-title>
          .
          <source>Bioinformatics</source>
          <volume>36</volume>
          (
          <issue>4</issue>
          ),
          <volume>1234</volume>
          {
          <fpage>1240</fpage>
          (
          <year>2020</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref24">
        <mixed-citation>
          24.
          <string-name>
            <surname>Lowe</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          , May eld, J.:
          <article-title>Extraction of reactions from patents using grammars</article-title>
          .
          <source>In: Working Notes of CLEF</source>
          <year>2020</year>
          |
          <article-title>Conference and Labs of the Evaluation Forum (</article-title>
          <year>2020</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref25">
        <mixed-citation>
          25.
          <string-name>
            <surname>Lowe</surname>
            ,
            <given-names>D.M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Sayle</surname>
            ,
            <given-names>R.A.</given-names>
          </string-name>
          :
          <article-title>LeadMine: a grammar and dictionary driven approach to entity recognition</article-title>
          .
          <source>Journal of cheminformatics 7(1)</source>
          , 1{
          <issue>9</issue>
          (
          <year>2015</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref26">
        <mixed-citation>
          26.
          <string-name>
            <surname>Ma</surname>
            ,
            <given-names>X.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Hovy</surname>
          </string-name>
          , E.:
          <article-title>End-to-end sequence labeling via bi-directional LSTM-CNNsCRF</article-title>
          .
          <source>In: Proceedings of the 54th Annual Meeting of the Association for Computational Linguistics</source>
          . pp.
          <volume>1064</volume>
          {
          <issue>1074</issue>
          (
          <year>2016</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref27">
        <mixed-citation>
          27.
          <string-name>
            <surname>Mahendran</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Gurdin</surname>
            ,
            <given-names>G.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Lewinski</surname>
            ,
            <given-names>N.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Tang</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>T.</surname>
          </string-name>
          ,
          <string-name>
            <surname>M.B.: NLPatVCU CLEF 2020 ChEMU Shared</surname>
          </string-name>
          <article-title>Task System Description</article-title>
          . In: Working Notes of CLEF 2020|
          <article-title>Conference and Labs of the Evaluation Forum (</article-title>
          <year>2020</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref28">
        <mixed-citation>
          28.
          <string-name>
            <surname>Mikolov</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Sutskever</surname>
            ,
            <given-names>I.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Chen</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Corrado</surname>
            ,
            <given-names>G.S.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Dean</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          :
          <article-title>Distributed representations of words and phrases and their compositionality</article-title>
          .
          <source>In: Advances in neural information processing systems</source>
          . pp.
          <volume>3111</volume>
          {
          <issue>3119</issue>
          (
          <year>2013</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref29">
        <mixed-citation>
          29.
          <string-name>
            <surname>Muresan</surname>
            ,
            <given-names>S.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Petrov</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Southan</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Kjellberg</surname>
            ,
            <given-names>M.J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Kogej</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Tyrchan</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Varkonyi</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Xie</surname>
            ,
            <given-names>P.H.</given-names>
          </string-name>
          :
          <article-title>Making every SAR point count: the development of Chemistry Connect for the large-scale integration of structure and bioactivity data</article-title>
          .
          <source>Drug Discovery Today</source>
          <volume>16</volume>
          (
          <fpage>23</fpage>
          -
          <lpage>24</lpage>
          ),
          <volume>1019</volume>
          {
          <fpage>1030</fpage>
          (
          <year>2011</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref30">
        <mixed-citation>
          30.
          <string-name>
            <surname>Nakov</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Hoogeveen</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Marquez</surname>
            ,
            <given-names>L.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Moschitti</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Mubarak</surname>
            ,
            <given-names>H.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Baldwin</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Verspoor</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          :
          <article-title>Semeval-2017 task 3: Community question answering</article-title>
          . arXiv preprint arXiv:
          <year>1912</year>
          .
          <volume>00730</volume>
          (
          <year>2019</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref31">
        <mixed-citation>
          31.
          <string-name>
            <surname>Ngai</surname>
            ,
            <given-names>G.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Florian</surname>
          </string-name>
          , R.:
          <article-title>Transformation-based learning in the fast lane</article-title>
          . In:
          <article-title>Proceedings of the second meeting of the North American Chapter of the Association for Computational Linguistics on Language technologies</article-title>
          . pp.
          <volume>1</volume>
          {
          <issue>8</issue>
          (
          <year>2001</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref32">
        <mixed-citation>
          32.
          <string-name>
            <surname>Nguyen</surname>
            ,
            <given-names>D.Q.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Zhai</surname>
            ,
            <given-names>Z.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Yoshikawa</surname>
            ,
            <given-names>H.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Fang</surname>
            ,
            <given-names>B.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Druckenbrodt</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Thorne</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Hoessel</surname>
            ,
            <given-names>R.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Akhondi</surname>
            ,
            <given-names>S.A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Cohn</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Baldwin</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Verspoor</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          :
          <article-title>ChEMU: Named Entity Recognition and Event Extraction of Chemical Reactions from Patents</article-title>
          .
          <source>In: Proceedings of the 42nd European Conference on Information Retrieval</source>
          . pp.
          <volume>572</volume>
          {
          <issue>579</issue>
          (
          <year>2020</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref33">
        <mixed-citation>
          33.
          <string-name>
            <surname>Palmer</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Gildea</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Kingsbury</surname>
            ,
            <given-names>P.:</given-names>
          </string-name>
          <article-title>The proposition bank: An annotated corpus of semantic roles</article-title>
          .
          <source>Computational Linguistics</source>
          <volume>31</volume>
          (
          <issue>1</issue>
          ),
          <volume>71</volume>
          {
          <fpage>106</fpage>
          (
          <year>2005</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref34">
        <mixed-citation>
          34.
          <string-name>
            <surname>Pattabhi</surname>
            ,
            <given-names>M.C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Rao</surname>
            <given-names>.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>R.</surname>
          </string-name>
          ,
          <string-name>
            <given-names>Lalitha</given-names>
            <surname>Devi</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.</given-names>
            :
            <surname>CLRG ChemNER: A Chemical Named Entity Recognizer @ ChEMU CLEF</surname>
          </string-name>
          <article-title>2020</article-title>
          . In: Working Notes of CLEF 2020|
          <article-title>Conference and Labs of the Evaluation Forum (</article-title>
          <year>2020</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref35">
        <mixed-citation>
          35.
          <string-name>
            <surname>Peters</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Neumann</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Iyyer</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Gardner</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Clark</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Lee</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Zettlemoyer</surname>
            ,
            <given-names>L.</given-names>
          </string-name>
          :
          <article-title>Deep contextualized word representations</article-title>
          .
          <source>In: Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics</source>
          . pp.
          <volume>2227</volume>
          {
          <issue>2237</issue>
          (
          <year>2018</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref36">
        <mixed-citation>
          36.
          <string-name>
            <surname>Radford</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Wu</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Child</surname>
            ,
            <given-names>R.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Luan</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Amodei</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Sutskever</surname>
            ,
            <given-names>I.</given-names>
          </string-name>
          :
          <article-title>Language models are unsupervised multitask learners</article-title>
          .
          <source>OpenAI Blog</source>
          <volume>1</volume>
          (
          <issue>8</issue>
          ),
          <volume>9</volume>
          (
          <year>2019</year>
          )
        </mixed-citation>
      </ref>
    </ref-list>
  </back>
</article>