<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.0 20120330//EN" "JATS-archivearticle1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta />
    <article-meta>
      <title-group>
        <article-title>Assessment of NER solutions against the first and second CALBC Silver Standard Corpus</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <string-name>Dietrich Rebholz-Schuhmann</string-name>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Antonio Jimeno Yepes</string-name>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Chen Li</string-name>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Senay Kafkas</string-name>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ian Lewin</string-name>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ning Kang</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Peter Corbett</string-name>
          <xref ref-type="aff" rid="aff11">11</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>David Milward</string-name>
          <xref ref-type="aff" rid="aff11">11</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ekaterina Buyko</string-name>
          <xref ref-type="aff" rid="aff10">10</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Elena Beisswanger</string-name>
          <xref ref-type="aff" rid="aff10">10</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Kerstin Hornbostel</string-name>
          <xref ref-type="aff" rid="aff10">10</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Alexandre Kouznetsov</string-name>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>René Witte</string-name>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Jonas B. Laurila</string-name>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Christopher JO Baker</string-name>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Cheng-Ju Kuo</string-name>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Simone Clematide</string-name>
          <xref ref-type="aff" rid="aff16">16</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Fabio Rinaldi</string-name>
          <xref ref-type="aff" rid="aff16">16</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Richárd Farkas</string-name>
          <xref ref-type="aff" rid="aff13">13</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>György Móra</string-name>
          <xref ref-type="aff" rid="aff13">13</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Kazuo Hara</string-name>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Laura Furlong</string-name>
          <xref ref-type="aff" rid="aff14">14</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Michael Rautschka</string-name>
          <xref ref-type="aff" rid="aff14">14</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Mariana Lara Neves</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Alberto Pascual-Montano</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Qi Wei</string-name>
          <xref ref-type="aff" rid="aff12">12</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Nigel Collier</string-name>
          <xref ref-type="aff" rid="aff12">12</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Md. Faisal Mahbub Chowdhury</string-name>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Alberto Lavelli</string-name>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Rafael Berlanga</string-name>
          <xref ref-type="aff" rid="aff15">15</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Roser Morante</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Vincent Van Asch</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Walter Daelemans</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>José Luís Marina</string-name>
          <xref ref-type="aff" rid="aff2">2</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Erik van Mulligen</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Jan Kors</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Udo Hahn</string-name>
          <xref ref-type="aff" rid="aff10">10</xref>
        </contrib>
        <aff id="aff0">
          <label>0</label>
          <institution>BioComputing Unit, National Center of Biotechnology (CNB-CSIC)</institution>
          ,
          <addr-line>Madrid</addr-line>
          ,
          <country country="ES">Spain</country>
        </aff>
        <aff id="aff1">
          <label>1</label>
          <institution>CLiPS, University of Antwerp</institution>
          ,
          <country country="BE">Belgium</country>
        </aff>
        <aff id="aff2">
          <label>2</label>
          <institution>Complutense University of Madrid</institution>
          ,
          <country country="ES">Spain</country>
        </aff>
        <aff id="aff3">
          <label>3</label>
          <institution>Dept. of Computer Science &amp; Applied Statistics, University of New Brunswick</institution>
          ,
          <country country="CA">Canada</country>
        </aff>
        <aff id="aff4">
          <label>4</label>
          <institution>Dept. of Computer Science &amp; Software Engineering, Concordia University</institution>
          ,
          <addr-line>Montreal</addr-line>
          ,
          <country country="CA">Canada</country>
        </aff>
        <aff id="aff5">
          <label>5</label>
          <institution>Dept. of Medical Informatics, Erasmus University Medical Center</institution>
          ,
          <addr-line>Rotterdam, NL</addr-line>
        </aff>
        <aff id="aff6">
          <label>6</label>
          <institution>EMBL Outstation, European Bioinformatics Institute</institution>
          ,
          <addr-line>Hinxton, Cambridge, CB10 1SD</addr-line>
          ,
          <country country="UK">U.K</country>
        </aff>
        <aff id="aff7">
          <label>7</label>
          <institution>Fondazione Bruno Kessler</institution>
          ,
          <addr-line>Trento</addr-line>
          ,
          <country country="IT">Italy</country>
        </aff>
        <aff id="aff8">
          <label>8</label>
          <institution>Institute of Information Science</institution>
          ,
          <addr-line>Academia Sinica, Taipei 115</addr-line>
          ,
          <country country="TW">Taiwan</country>
        </aff>
        <aff id="aff9">
          <label>9</label>
          <institution>Institute of Science and Technology</institution>
          ,
          <addr-line>Nara</addr-line>
          ,
          <country country="JP">Japan</country>
        </aff>
        <aff id="aff10">
          <label>10</label>
          <institution>Language &amp; Information Engineering (JULIE) Lab, Friedrich-Schiller-Universität</institution>
          ,
          <addr-line>Jena</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff11">
          <label>11</label>
          <institution>Linguamatics Ltd, St. John's Innovation Centre</institution>
          ,
          <addr-line>Cambridge</addr-line>
          ,
          <country country="UK">U.K</country>
        </aff>
        <aff id="aff12">
          <label>12</label>
          <institution>National Institute of Informatics</institution>
          ,
          <addr-line>Tokyo</addr-line>
          ,
          <country country="JP">Japan</country>
        </aff>
        <aff id="aff13">
          <label>13</label>
          <institution>Research Group on Artificial Intelligence, Hungarian Academy of Sciences</institution>
          ,
          <country country="HU">Hungary</country>
        </aff>
        <aff id="aff14">
          <label>14</label>
          <institution>Research Programme on Biomedical Informatics (GRIB) IMIM, DCEX, Universitat Pompeu Fabra</institution>
          ,
          <addr-line>Barcelona</addr-line>
          ,
          <country country="ES">Spain</country>
        </aff>
        <aff id="aff15">
          <label>15</label>
          <institution>Universitat Jaume I</institution>
          ,
          <country country="ES">Spain</country>
        </aff>
        <aff id="aff16">
          <label>16</label>
          <institution>University of Zürich</institution>
          ,
          <country country="CH">Switzerland</country>
        </aff>
      </contrib-group>
      <fpage>63</fpage>
      <lpage>71</lpage>
      <abstract>
        <p>Background Text mining challenges have been organised to measure the performance of automatic text mining solutions against a manually annotated gold standard corpus (GSC). The preparation of the GSC is timeconsuming and costly and the final corpus consists at the most of a few thousand documents annotated with a limited set of semantic groups. To overcome these shortcomings, the CALBC project partners (PPs) have produced a large-scale annotated biomedical corpus with four different semantic groups through the harmonisation of annotations from automatic text mining solutions, the first version of the Silver Standard Corpus (SSC-I). The four semantic groups were chemical entities and drugs (CHED), genes and proteins (PRGE), diseases and disorders (DISO) and species (SPE). This corpus has been used for the First CALBC Challenge asking the participants to annotate the corpus with their annotation solutions.</p>
      </abstract>
    </article-meta>
  </front>
  <body>
    <sec id="sec-1">
      <title>Results</title>
      <p>All four PPs from the CALBC project and in addition, 12 challenge participants (CPs) contributed
annotated data sets for evaluation against the SSC-I. CPs could ignore the training data and deliver the
annotations from their annotation system, or could train a machine-learning approach on the provided
preannotated data. In general, the performances of the annotation solutions were lower for the CHED and
PRGE in comparison to the identification of DISO and SPE. The best performance over all semantic
groups were achieved from two annotation solutions that have been trained on the SSC-I.
The data sets from participants were used to generate the harmonised Silver Standard Corpus II (SSC-II), if
the participant did not make use of the annotated data set from the SSC-I for training purposes. The
performances of the participants’ solutions were again measured against the SSC-II. The performances of
the annotation solutions showed again better results for DISO and SPE in comparison to CHED and PRGE.</p>
    </sec>
    <sec id="sec-2">
      <title>Conclusions</title>
      <p>The SSC-I delivers a large set of annotations (1,121,705) for a large number of documents (100,000
Medline abstracts). The annotations cover four different semantic groups and are sufficiently homogeneous
to be reproduced with a trained classifier leading to an average F-measure of 85%. Benchmarking the
annotation solutions against the SSC-II leads to better performance for the CPs’ annotation solutions in
comparison to the SSC-I.</p>
      <sec id="sec-2-1">
        <title>Background</title>
        <p>
          Biomedical text mining (TM) has developed into a bioinformatics discipline seeking IT methods delivering
accurate results from an automatic literature analysis into bioinformatics research. This objective requires
the development of benchmarks and the assessment of existing TM solutions against these benchmarks. A
number of challenges have been proposed to achieve this goal: BioCreAtive I and II, JNLPBA and others
[
          <xref ref-type="bibr" rid="ref1 ref2 ref3 ref4 ref5">1-5</xref>
          ]. In all these approaches, the organisers deliver a set of manually annotated documents and ask the
CPs to reproduce the results with their automatic methods. The annotated corpora are provided to the
public after the challenge is closed and all the results are published.
        </p>
        <p>
          The first CALBC Challenge is similar in the sense that the project partners (PPs) of the CALBC project
also provided an annotated corpus to the challenge participants (CPs) of the first CALBC challenge to
reproduce the annotations with automatic means. On the other side, the first CALBC Challenge was
different to the before-mentioned challenges with regards to the following modifications: (1) the annotated
corpus has been generated automatically and not manually (Silver Standard Corpus, SSC-I), and (2) the
size of the SSC-I is significantly bigger than the corpora mentioned produced for the other challenges, i.e.
the annotated corpus contains 50,000 Medline abstracts for training and the corpus for annotation consists
of 100,000 test documents. This difference in size requires that all assessment is performed fully
automatically, that the CPs apply annotation solutions that can cope with such a large-scale corpus and that
the assessment solutions can evaluate the contributions in a short period of time. The automatic annotation
of the corpus also requires new solutions to harmonise the contributions from different automatic
annotation solutions. The term “harmonisation” refers to the process that determines a set of annotation
boundaries in the text. Overall these annotations should have the characteristic that all annotation solutions
show high performance against the set of annotations, for example when measuring the F-measure of the
annotation solution [
          <xref ref-type="bibr" rid="ref6">6</xref>
          ].
        </p>
        <p>
          When comparing different NER solutions, it becomes clear that they do not generate the same results
depending on the technology behind and the type of resources used for the instantiation of the solutions
(see BioCreative II). On the other side, when combining the results from different automatic annotation
solutions, we can achieve an improvement of the results of the combined solution (see BioCreative
MetaServer) [
          <xref ref-type="bibr" rid="ref7">7</xref>
          ]. As a consequence, the PPs of the CALBC project have combined their automatic annotation
solutions to produce the first Silver Standard Corpus of the CALBC project.
        </p>
        <p>In addition, each annotation solution is optimised for a single semantic type and solutions for a larger
scope of semantic groups are still missing. This is again partly due to the fact that manually curated
corpora can only cover a small number of semantic groups to focus the ongoing work to the amount of
work that is achievable in a fixed period of time and according to the available budget. The proposed
approach of the CALBC project can cover a larger number of annotations due to the fact that the
annotations are produced automatically and harmonised with automatic means.</p>
        <p>In this manuscript, we report on the results of the first CALBC challenge. The CPs have submitted one or
several sets of annotated documents. All the submissions have been assessed against the SSC-I. In
addition, the submissions have been used to generate the second Silver Standard Corpus (SSC-II) and all
the submissions have been assessed against the SSC-II. The results are presented in this manuscript to
support a better understanding to which extent the automatic generation of an annotated corpus contributes
to the benchmarking of annotation solutions in a domain where a large number of NERs have to be
identified inside a large number of scientific documents.
In the CALBC project and challenge the PPs and CPs contribute their annotations on a given corpus to
enable the harmonisation of all annotations for a large-scale annotated corpus. A priori the annotation
solutions do not share any properties and the contributed annotations should be produced by independent
systems, but should be similar in the sense that they contribute annotations for entities in the biomedical
domain. This leads to the result that the different solutions make use of similar biomedical data resources
for the representation of terms and concepts and thus should show similarities in the annotation</p>
        <p>Solution
Dictionary-based
concept recognition
Indexing of tokens</p>
        <p>and terms
Both, trained &amp; rule- P03
based solutions
Case-based
reasoning
CRF based, trained</p>
        <p>NER solution</p>
        <p>P02
P04
P06
P10
P13
P15
P09
P07
P16
P11
P12
P14</p>
        <p>Use of
PPs | Training
CPs Data
P01 [ / ]</p>
      </sec>
    </sec>
    <sec id="sec-3">
      <title>Generation of the first CALBC Silver Standard Corpus (SSC-I)</title>
      <p>
        All PPs annotated the corpus of 150,000 Medline abstracts with their annotation solutions. The project
partners P01, P02 and P04 used dictionary-based concept recognition methods with techniques for quality
improvements, whereas partner P03 applied a combination of solutions that are either dictionary-based or
is based on machine-learning techniques. All annotations were delivered in the IeXML format and concept
normalisation should make use of standard resources such as UMLS, UniProtKb, EntrezGene or should
follow the UMLS semantic type system [
        <xref ref-type="bibr" rid="ref10 ref11 ref12 ref13 ref9">9-13</xref>
        ].
      </p>
      <p>
        The alignment is based on the methods described in [
        <xref ref-type="bibr" rid="ref14 ref6">6,14</xref>
        ]. The applied method used pair-wise alignment
between two annotated sets for a given semantic type. For every sentence the annotations from one
contribution for a given type is aligned with the annotations from the next contribution for the same
semantic type. The tokens have been weighted with the inverse document frequency (IDF) for the token
across the whole corpus and the cosine similarity of the two annotations has been measured. If the
similarity is above 0.98, then the alignment is considered successful and the boundaries of the shorter
annotation have been selected as the final annotation. If at least two partner contributions agree on the
same annotation (2-vote agreement), then the annotation has been selected for the final corpus. Only in the
case of CHED the PPs shared the terminological resource for the annotation [
        <xref ref-type="bibr" rid="ref15">15</xref>
        ].
      </p>
    </sec>
    <sec id="sec-4">
      <title>Generation of the SSC-II</title>
      <p>
        The contributions of the CPs were evaluated against the SSC-I. Different evaluation schemes were used to
determine the performance of the solutions [
        <xref ref-type="bibr" rid="ref14 ref6">6,14</xref>
        ]. All contributions were assessed against the SSC-I by
applying exact matching, nested matching and cosine similarity matching with a 0.98 and 0.9 cosine
similarity score (results not shown). The measurements were performed on the basis of a reduced but
standardised set of 1,000 Medline abstracts that have been selected at random from the full corpus.
CHED
PRGE
DISO
SPE
anNntro.aOtiofns Nr. Of Anvre.rOagfe anNnort.aOtiofns
in the SSC- Nr. Of CPs submissions annotations in the
SSC
      </p>
      <p>I from CPs frComPsall II
228,622 6 11 233,398 238,431
275,235 9 15 343,681 435,797
300,637 8 11 255,599 245,524
317,211 7 9 277,071 304,503</p>
      <p>The best average F-measure performances were achieved when applying 0.9 and 0.98 cosine similarity
scoring. All submissions from all participants have been evaluated and the contributions with the best
Fmeasure performance against the SSC-I have been selected for the harmonisation into the SSC-II.
For the harmonisation of the contributions (SSC-II), a varying number of contributions had to be
considered for the different semantic groups, i.e. 6 for CHED, 7 for SPE, 8 for DISO and 9 for PRGE (see
table 2). A 3-vote agreement in combination with a 0.98 cosine similarity score in the alignment was
required for the acceptance of the annotation across the different contributions. For all semantic groups,
different voting schemes (i.e. 2- to 6-vote agreement) were evaluated to determine the best performing
voting scheme in terms of the highest average F-measure performance across the contributions of the CPs.
The 3-vote agreement delivered the best balance between the recall and the precision of the contributions
against the harmonised SSC-II. All presented results are based on the annotations on the subset of 1,000
Medline abstracts.</p>
      <p>The alignments of the 100,000 documents were either performed on Sun Fire opteron servers (4 or 8 CPUs,
RAM sizes from 32 to 256 Gb RAM, 9-12 hours) or on the compute farm of 700 IBM compute engines
(dual CPU, 1.2-2.8 Ghz, 2 GB RAM, 3 hours).</p>
    </sec>
    <sec id="sec-5">
      <title>Challenge participation and challenge contributions</title>
      <p>12 CPs actively took part in the challenge. Each CP could contribute several submissions at any time.
Overall the CALBC challenge received 19 valid submissions. 3 CPs used the SSC-I as training data and
contributed in total 8 submissions (ref. to table 2). 2 CPs did not use the SSC-I, but used an annotation
solution that has been trained on a different annotated corpus for the challenge. All other partners (in total
11) used dictionary-based solutions and in one case used a combination of different solutions without
training on the SSC-I.</p>
      <p>Five CPs only focused on a single semantic group. All the other CPs covered three or more semantic
groups. CP P10 delivered for PRGE a very high number of annotations, which impaired the performance
of the system against the SSC-I.</p>
      <sec id="sec-5-1">
        <title>Results</title>
        <p>The PPs contributions have been aligned to generated the SSC-I. The SSC-I has been contributed to the
public to train machine-learning based NER solutions on the corpus and to gather the annotations of the
CPs for performance assessments. The contributions of the CPs have been used to generate the SSC-II.</p>
      </sec>
    </sec>
    <sec id="sec-6">
      <title>Evaluation of the contributed annotated corpora against the SSC-I</title>
      <p>
        The submissions of the CPs were compared against the SSC-I (see below). Table 3 shows that two
solutions that were based on the provided training data reproduced the SSC-I annotation „standard“ at a
high level of quality: for SPE the solutions achieved 93% F-measure and for the other semantic groups the
F-measure was above 80% [
        <xref ref-type="bibr" rid="ref8">8</xref>
        ]. This shows that the SSC-I was homogeneous enough so that a trained
system could reproduce the annotations. As a consequence, we can expect that complex automatic
annotation solutions could be replaced with a machine learning approach to reproduce the annotations.
80.0% 100.0%
      </p>
      <p>
        Precision
The two best-performing machine-learning based solutions produce results that are comparable to known
solutions for the gene mention task [
        <xref ref-type="bibr" rid="ref16 ref17">16,17</xref>
        ]. On the other side, the performances have been measured
against a corpus that includes a higher degree of variability in the annotations in comparison to the gold
standard corpora that are usually used for the measurement of gene-tagging solutions.
      </p>
      <p>Fig. 2 shows a distribution for the performance for the annotation for chemical entities. The two best
performing machine-learning solutions outperform again all other annotation solutions and the PPs’
annotation solutions have performances that are rather similar to each other and quite different from the
performances of the contributions from the CPs. Comparing the results in fig. 1 and fig. 2, we note that the
best annotation solutions show better performance for chemical entities than the same solutions
67
demonstrate for the annotation of proteins/genes. We conclude that the annotation of PRGEs in the SSC-I
have higher variability (or more noise) than the annotations for the chemical entities in the same corpora.
The following figures (fig. 3 and 4) show the same distribution for disease and species mention
identification. For these two tasks the annotation solutions show better performance than for the previous
two tasks (PRGE and CHED). We can derive that a good performance on these two tasks can be reached
by the majority of the annotation solutions in comparison to the other two tasks.</p>
      <p>20.0%
40.0%
60.0%
80.0% 100.0%</p>
      <p>Precision
20.0%
40.0%
60.0%
80.0% 100.0%</p>
      <p>Precision</p>
      <p>The diagram for the identification of the diseases (DISO, fig. 3) demonstrates that the majority of the
proposed systems identified the diseases at a recall of 60% and above, and at a precision of 55% and
above. Two rule-based solutions from CPs showed similar performances to the PP’s solutions. We can
conclude that the representation of the diseases in the SSC-I is better standardised and thus includes less
variability or noise than the representation of proteins/genes and chemical entities.</p>
      <p>The identification of species could be solved to the best precision and the best recall values from the large
majority of all proposed solutions. Again the two best performances were achieved by two
machinelearning approaches that reproduced the annotations from the training data. The performances of the other
solutions, i.e. the PPs’ solutions and the CPs’ solutions, had the best performances for the identification of
species in contrast to the other tasks. It is clear that the identification of species can be performed at a level
of quality which is above the measured performances of the other semantic groups.</p>
    </sec>
    <sec id="sec-7">
      <title>Performance against the SSC-II</title>
      <p>SPE
DISO
CHEM
PRGE
The results of the CPs and the PPs were compared against the SSC-II in addition to the SSC-I. Both
harmonised sets were generated by using 98% cosine similarity and the comparison against the corpus was
done with the same measure.</p>
      <p>Tagging of proteins/genes and chemical entities measured against the SSC-II
The performances of the PPs’ annotation solutions for genes/proteins showed lower results in the
assessment against the SSC-II than in comparison to the SSC-I (refer to fig. 1). Since the SSC-II represents
the harmonisation of annotations across a larger number of contributions, it can be expected that the
annotations in the SSC-II are more heterogeneous than in the SSC-I.</p>
      <p>The performance of the CPs’ annotation solutions has improved against the SSC-II in comparison to the
SSC-I: the precision against the SSC-II has increased in comparison to the SSC-I. Recall has also
improved. This result shows that the SSC-II incorporates characteristic features that are shared amongst all
annotation solutions.</p>
      <p>In the SSC-II the annotation solutions of the PPs for chemical entities show lower performance in
comparison to the SSC-I (refer to fig. 2). The performance of the CPs’ annotation solutions has improved.
Altogether the distribution of the performances of the PPs’ annotation solutions and the CPs’ solutions is
comparable.</p>
      <p>SSC-I / PRGE</p>
      <p>SSC-II / PRGE</p>
      <p>SSC-I / CHEM</p>
      <p>SSC-II / CHEM
0</p>
      <p>P13 P24 P3 1 P4 2 P510 P68 P175 P89 P96 10
0</p>
      <p>P13</p>
      <p>P24</p>
      <p>P31</p>
      <p>P42</p>
      <p>P510 P68</p>
      <p>P715
8</p>
      <p>As can be seen in fig. 5, the performances of the annotation solutions for the four PPs deteriorated when
comparing the performance against SSC-II instead of SSC-I. Furthermore, the performance of the four PPs
against the SSC-II shows an F-measure that seems to be more evenly distributed across the different PPs,
i.e. the systems seem to be more equal.</p>
      <p>The performances of the CPs’ annotation solutions have improved when moving from the SSC-I to the
SSC-II. This result can be explained by the fact that the contributions of the CPs have been included into
the SSC-II in comparison to the SSC-I.</p>
      <p>The results from the comparison of the annotation solutions for the chemical entities are not as clear as the
results for the annotation of proteins/genes. In the case of the chemical entities, the performances of the
PPs’ solutions deteriorate except for one PP. The performance of the CPs’ solutions varies to a small
extent.</p>
      <p>Tagging of diseases and species measured against the SSC-II
The PPs’ annotation solutions and the CPs’ solutions show similar performance against the SSC-II and the
SSC-I. The two corpora seem to have the same characteristics concerning the annotated entities in the
corpus. In other words, the contribution of the CPs to the harmonised corpus did not change the quality of
the silver standard corpus when producing the SSC-II from the PPs’ and the CPs’ contributions in
comparison to the SSC-I. We can conclude that the annotation of disease entities is better normalised than
the two other semantic groups, i.e. chemical entities and protein/genes, respectively.</p>
      <p>Similar to the assessment of disease annotations, the species tagging solutions of the PPs and the project
CPs did not vary when the annotations were evaluated against the SSC-II in comparison to the SSC-I. For
both corpora, the annotation solutions yielded similar results. This leads to the conclusion that the SSC-I
and the SSC-II have similar annotations and also to the result that the different contributing systems had
100%
90%
80%
70%
60%
50%
40%
30%
20%
10%
0%
similar performances right from the beginning. Overall, we can conclude that the representation of species
is better normalised or standardised in the scientific literature than chemical entities or gene/protein
representations.</p>
      <p>In the last analysis, we compared the F-measures reached from the individual systems against the SSC-II
directly against the F-measures from the SSC-I. This should give an overview on the solutions that gained
performance in the SSC-II over the SSC-I and the other solutions that deteriorated their performance.</p>
      <p>SSC-I / SPE SSC-II / SPE SSC-I / DISO SSC-II / DISO
0</p>
      <p>P13 P24</p>
      <p>P3 1 P42 P510 P68 P715 P89
9
0</p>
      <p>P13 P24 P31 P4 2 P510 P68 P715 P89 P96 10</p>
      <p>When analysing the performance of the different solutions for species annotations and for diseases
annotations, we find only small differences in the performances of the systems against the SSC-I and the
SSC-II.</p>
      <p>Direct measurement of the SSC-I against the SSC-II</p>
      <p>Reference: SSC-I (cos 0.98)
SSC-II DISO SPE PRGE CHED
Rec 89.0% 94.5% 59.7% 49.6%
Prec 71.6% 90.0% 96.8% 49.4%</p>
      <p>F-meas 79.3% 92.2% 73.8% 49.5%</p>
      <p>In the direct comparison between the SSC-I and the SSC-II, the annotations for SPE and DISO
show better agreement than the comparison of the annotations for PRGE and CHED. The latter
shows the lowest performance indicating that higher diversity exists between the two corpora.</p>
      <sec id="sec-7-1">
        <title>Discussion &amp; Conclusions</title>
        <p>Manual inspection of the SSC-I and the SSC-II
The manual analysis of the SSC-I and the SSC-II is ongoing work. Due to the size of the corpus, it
requires special IT solutions to oversee the regularities and irregularities in the corpus. A selection of
irregularities result from the methods applied. First, a number of annotations are not captured (“false
negatives”, FN, reduced recall) if none of the solutions identifies the entities. An increasing number of
contributing annotation solutions reduces the risk that annotations are missed: a bigger number of included
annotation solutions lead to a bigger number of annotations that are captured. This achievement is
counterbalanced by the number of agreements that have to be available at minimum to accept an annotation.
Second, for the same type of entity, e.g. “insulin”, different annotation solutions use a different tag, e.g.
PRGE instead of CHED and vice versa. The harmonisation of the corpus can account for this, but will not
produce this type of polysemous annotation throughout the whole corpus, since not all mentions have been
consistently annotated with the two different groups over the whole corpus.</p>
        <p>Third, inflections of terms, e.g. “tumour” vs. “tumours” and “bear” vs. “bearing”, lead to disagreements
between the different annotation solutions. In the first case, the inflectional variability could be resolved
and would lead to higher agreement, in the second case assumptions about the usage of the verb or noun
have to be made to resolve conflicts.
The comparison of the proposed solutions against the SSC-I is a new approach to evaluate annotation
solutions. Until now, no large-scale corpus was available to achieve this task. In addition, it became clear
that the SSC-I is homogeneous enough to be used as training data to achieve the same annotation task
across the different semantic groups.</p>
        <p>The generation of a harmonised corpus is a challenging task, but the presented results demonstrate that the
produced harmonised corpus integrates the characteristics from the different annotation solutions. As a
result, we can determine the features in the harmonised corpus by the annotation solutions that contribute
to the generation of the SSC.</p>
        <p>From a different perspective, we can argue that each of the used annotation solutions represents a piece of
the complete annotation task. The more solutions are combined, the more closely we approximate an
assumed consensus in the annotation task, which can be reproduced with a machine-learning tagging
solution.</p>
      </sec>
      <sec id="sec-7-2">
        <title>References</title>
      </sec>
    </sec>
  </body>
  <back>
    <ref-list>
      <ref id="ref1">
        <mixed-citation>
          1.
          <string-name>
            <surname>Hirschman</surname>
            ,
            <given-names>L.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Yeh</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Blaschke</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Valencia</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          (
          <year>2005</year>
          ).
          <article-title>Overview of BioCreAtIvE: Critical assessment of information extraction for biology</article-title>
          .
          <source>BMC Bioinformatics</source>
          ,
          <volume>6</volume>
          (
          <issue>Suppl 1</issue>
          ),
          <fpage>S1</fpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref2">
        <mixed-citation>
          2.
          <string-name>
            <surname>Krallinger</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Morgan</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Smith</surname>
            ,
            <given-names>L.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Leitner</surname>
            ,
            <given-names>F.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Ta-nabe</surname>
          </string-name>
          , L.,
          <string-name>
            <surname>Wilbur</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Hirschman</surname>
            ,
            <given-names>L.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Valencia</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          (
          <year>2008</year>
          ).
          <article-title>Evaluation of textmining systems for biology: Overview of the Second BioCreAtIvE Community Challenge</article-title>
          .
          <source>Genome Biology</source>
          ,
          <volume>9</volume>
          (
          <issue>Suppl 2</issue>
          ),
          <fpage>S1</fpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref3">
        <mixed-citation>
          3.
          <string-name>
            <surname>Kim</surname>
            ,
            <given-names>J.D.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Ohta</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Tsuruoka</surname>
            ,
            <given-names>Y.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Tateishi</surname>
            ,
            <given-names>Y.</given-names>
          </string-name>
          , and
          <string-name>
            <surname>Collier</surname>
            ,
            <given-names>N.</given-names>
          </string-name>
          (
          <year>2004</year>
          ).
          <article-title>Introduction to the bio-entity recognition task at JNLPBA</article-title>
          .
          <source>In Proceedings of the JNLPBA-04</source>
          ,
          <fpage>70</fpage>
          -
          <lpage>75</lpage>
          , Geneva, Switzerland.
        </mixed-citation>
      </ref>
      <ref id="ref4">
        <mixed-citation>
          4.
          <string-name>
            <surname>Kim</surname>
            ,
            <given-names>J.D.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Ohta</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Pyysalo</surname>
            ,
            <given-names>S.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Kano</surname>
            ,
            <given-names>Y.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Tsujii</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          (
          <year>2009</year>
          ).
          <article-title>Overview of BioNLP'09 Shared Task on Event Extraction</article-title>
          .
          <source>In Proceedings of the Workshop on BioNLP: Shared Task</source>
          ,
          <fpage>1</fpage>
          -
          <lpage>9</lpage>
          , Colorado, USA.
        </mixed-citation>
      </ref>
      <ref id="ref5">
        <mixed-citation>5. LLL'05 challenge: http://www.cs.york.ac.uk/aig/lll/lll05/</mixed-citation>
      </ref>
      <ref id="ref6">
        <mixed-citation>
          6.
          <string-name>
            <surname>Rebholz-Schuhmann</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          ,
          <string-name>
            <given-names>A.J. Jimeno</given-names>
            <surname>Yepes</surname>
          </string-name>
          ,
          <string-name>
            <surname>E.M. van Mulligen</surname>
          </string-name>
          ,
          <string-name>
            <given-names>N.</given-names>
            <surname>Kang</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Kors</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Milward</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Corbett</surname>
          </string-name>
          ,
          <string-name>
            <given-names>E.</given-names>
            <surname>Buyko</surname>
          </string-name>
          ,
          <string-name>
            <given-names>K.</given-names>
            <surname>Tomanek</surname>
          </string-name>
          , E. Beisswanger, and
          <string-name>
            <given-names>U.</given-names>
            <surname>Hahn</surname>
          </string-name>
          .
          <article-title>(2010b) The CALBC Silver Standard Corpus for Biomedical Named Entities: A Study in Harmonizing the Contributions from Four Independent Named Entity Taggers</article-title>
          .
          <source>Proc. LREC</source>
          <year>2010</year>
          , ELRA, Valletta, Malta.
        </mixed-citation>
      </ref>
      <ref id="ref7">
        <mixed-citation>
          7.
          <string-name>
            <surname>Leitner</surname>
            ,
            <given-names>F.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Krallinger</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Rodriguez-Penagos</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Hakenberg</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Plake</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Kuo</surname>
            ,
            <given-names>C.J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Hsu</surname>
            ,
            <given-names>C.N.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Tsai</surname>
          </string-name>
          , R.T.,
          <string-name>
            <surname>Hung</surname>
            ,
            <given-names>H.C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Lau</surname>
            ,
            <given-names>W.W.</given-names>
          </string-name>
          , Johnson,
          <string-name>
            <given-names>C.A.</given-names>
            ,
            <surname>Saetre</surname>
          </string-name>
          ,
          <string-name>
            <given-names>R.</given-names>
            ,
            <surname>Yoshida</surname>
          </string-name>
          ,
          <string-name>
            <given-names>K.</given-names>
            ,
            <surname>Chen</surname>
          </string-name>
          ,
          <string-name>
            <given-names>Y.H.</given-names>
            ,
            <surname>Kim</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.</given-names>
            ,
            <surname>Shin</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.Y.</given-names>
            ,
            <surname>Zhang</surname>
          </string-name>
          ,
          <string-name>
            <given-names>B.T.</given-names>
            ,
            <surname>Baumgartner</surname>
          </string-name>
          ,
          <string-name>
            <given-names>W.A.</given-names>
            <surname>Jr</surname>
          </string-name>
          ,
          <string-name>
            <surname>Hunter</surname>
            ,
            <given-names>L.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Haddow</surname>
            ,
            <given-names>B.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Matthews</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Wang</surname>
            ,
            <given-names>X.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Ruch</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Ehrler</surname>
            ,
            <given-names>F.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Ozgür</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Erkan</surname>
            ,
            <given-names>G.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Radev</surname>
            ,
            <given-names>D.R.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Krauthammer</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Luong</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Hoffmann</surname>
            ,
            <given-names>R.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Sander</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Valencia</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          <article-title>Introducing metaservices for biomedical information extraction</article-title>
          .
          <source>Genome Biol</source>
          .
          <year>2008</year>
          ;
          <volume>9</volume>
          <issue>Suppl 2</issue>
          :
          <fpage>S6</fpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref8">
        <mixed-citation>
          8.
          <source>Proceedings of the First CALBC Workshop</source>
          (http://www.ebi.ac.uk/Rebholz-srv/CALBC/docs/FirstProceedings.pdf)
        </mixed-citation>
      </ref>
      <ref id="ref9">
        <mixed-citation>
          9.
          <string-name>
            <surname>Rebholz-Schuhmann</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Kirsch</surname>
            ,
            <given-names>H.</given-names>
          </string-name>
          , and
          <string-name>
            <surname>Nenadic</surname>
            ,
            <given-names>G.</given-names>
          </string-name>
          (
          <year>2006</year>
          )
          <article-title>IeXML: towards a framework for interoperability of text processing modules to improve annotation of semantic types in biomedical text</article-title>
          .
          <source>Proc. of BioLINK</source>
          , ISMB 2006, Fortaleza, Brazil.
        </mixed-citation>
      </ref>
      <ref id="ref10">
        <mixed-citation>
          10.
          <string-name>
            <given-names>O.</given-names>
            <surname>Bodenreider</surname>
          </string-name>
          and
          <string-name>
            <given-names>A.</given-names>
            <surname>McCray</surname>
          </string-name>
          ,
          <year>2003</year>
          ,
          <article-title>Exploring semantic groups through visual approaches</article-title>
          ,
          <source>Journal of Biomedical Informatics</source>
          <volume>36</volume>
          (
          <issue>6</issue>
          ):
          <fpage>414</fpage>
          -
          <lpage>432</lpage>
          ,
          <year>2003</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref11">
        <mixed-citation>
          11.
          <string-name>
            <surname>Bodenreider</surname>
            <given-names>O</given-names>
          </string-name>
          :
          <article-title>The Unified Medical Language System (UMLS): integrating biomedical terminology</article-title>
          .
          <source>Nucleic Acids Res</source>
          <year>2004</year>
          ,
          <volume>32</volume>
          (Database issue):
          <fpage>D267</fpage>
          -
          <lpage>270</lpage>
        </mixed-citation>
      </ref>
      <ref id="ref12">
        <mixed-citation>
          12.
          <article-title>The Universal Protein Resource (UniProt) 2009</article-title>
          .
          <source>Nucleic Acids Res</source>
          <year>2009</year>
          ,
          <volume>37</volume>
          (Database issue):
          <fpage>D169</fpage>
          -
          <lpage>174</lpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref13">
        <mixed-citation>
          13.
          <string-name>
            <surname>Maglott</surname>
            <given-names>D</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Ostell</surname>
            <given-names>J</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Pruitt</surname>
            <given-names>KD</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Tatusova</surname>
            <given-names>T</given-names>
          </string-name>
          :
          <article-title>Entrez Gene: gene-centered information at NCBI</article-title>
          .
          <source>Nucleic Acids Res</source>
          <year>2007</year>
          ,
          <volume>35</volume>
          (Database issue):
          <fpage>D26</fpage>
          -
          <lpage>31</lpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref14">
        <mixed-citation>
          14.
          <string-name>
            <surname>Rebholz-Schuhmann</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          ,
          <string-name>
            <given-names>A. Jimeno</given-names>
            <surname>Yepes</surname>
          </string-name>
          ,
          <string-name>
            <surname>E. Van Mulligen</surname>
          </string-name>
          ,
          <string-name>
            <given-names>N.</given-names>
            <surname>Kang</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Kors</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Milward</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Corbett</surname>
          </string-name>
          ,
          <string-name>
            <given-names>E.</given-names>
            <surname>Buyko</surname>
          </string-name>
          , E. Beisswanger, and
          <string-name>
            <given-names>U.</given-names>
            <surname>Hahn.</surname>
          </string-name>
          (
          <year>2010a</year>
          )
          <article-title>"CALBC Silver Standard Corpus."</article-title>
          <source>J Bioinform Comput Biol</source>
          .
          <source>2010 Feb;8</source>
          (
          <issue>1</issue>
          ):
          <fpage>163</fpage>
          -
          <lpage>79</lpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref15">
        <mixed-citation>
          15.
          <string-name>
            <surname>Hettne</surname>
            ,
            <given-names>KM</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Stierum</surname>
            <given-names>RH</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Schuemie</surname>
            <given-names>MJ</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Hendriksen</surname>
            <given-names>PJ</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Schijvenaars</surname>
            <given-names>BJ</given-names>
          </string-name>
          ,
          <string-name>
            <surname>van Mulligen</surname>
            <given-names>EM</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Kleinjans</surname>
            <given-names>J</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Kors</surname>
            <given-names>JA.</given-names>
          </string-name>
          <article-title>A dictionary to identify small molecules and drugs in free text</article-title>
          .
          <source>Bioinformatics</source>
          <year>2009</year>
          ;
          <volume>25</volume>
          :
          <fpage>2983</fpage>
          -
          <lpage>91</lpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref16">
        <mixed-citation>
          16.
          <string-name>
            <surname>Leaman</surname>
            ,
            <given-names>R.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Gonzalez</surname>
            ,
            <given-names>G.</given-names>
          </string-name>
          (
          <year>2008</year>
          ).
          <article-title>BANNER: An executable survey of advances in biomedical named entity recognition</article-title>
          .
          <source>In Proceedings of the Pacific Symposium on Biocomputing</source>
          ,
          <volume>13</volume>
          :
          <fpage>652</fpage>
          -
          <lpage>663</lpage>
          , Hawaii.
        </mixed-citation>
      </ref>
      <ref id="ref17">
        <mixed-citation>
          17.
          <string-name>
            <surname>Torii</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Hu</surname>
            .
            <given-names>Z.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Wu</surname>
            <given-names>C.H.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Liu</surname>
            <given-names>H.</given-names>
          </string-name>
          (
          <year>2009</year>
          ).
          <article-title>BioTagger-GM: a gene/protein name recognition system</article-title>
          .
          <source>J Am Med Inform</source>
          ,
          <volume>16</volume>
          :
          <fpage>247</fpage>
          -
          <lpage>255</lpage>
          .
        </mixed-citation>
      </ref>
    </ref-list>
  </back>
</article>