<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.0 20120330//EN" "JATS-archivearticle1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta />
    <article-meta>
      <title-group>
        <article-title>Text mining and expert curation to develop a database on psychiatric diseases and their genes</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <string-name>Alba Gutie´rrez-Sacrist a´n GRIB IMIM - UPF</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
          <xref ref-type="aff" rid="aff4">4</xref>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Olga Valverde GReNeC IMIM - UPF</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
          <xref ref-type="aff" rid="aff4">4</xref>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Adriana Farre´ Parc de Salut Mar UAB</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
          <xref ref-type="aff" rid="aff4">4</xref>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Miguel A. Mayer GRIB IMIM - UPF</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
          <xref ref-type="aff" rid="aff4">4</xref>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Angela Leis GRIB IMIM - UPF</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
          <xref ref-type="aff" rid="aff4">4</xref>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Sandra Montagud-Romero</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
          <xref ref-type="aff" rid="aff4">4</xref>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <aff id="aff0">
          <label>0</label>
          <institution>Antonio Armario Universitat Auto` noma de Barcelona</institution>
        </aff>
        <aff id="aff1">
          <label>1</label>
          <institution>Francisco Javier Pavo ́ n Instituto de Investigacio ́ n Biome ́dica de Ma ́laga</institution>
        </aff>
        <aff id="aff2">
          <label>2</label>
          <institution>Jes u ́s Giraldo Universitat Aut o`noma de Barcelona</institution>
        </aff>
        <aff id="aff3">
          <label>3</label>
          <institution>Jordi Ortiz School of Medicine UAB</institution>
        </aff>
        <aff id="aff4">
          <label>4</label>
          <institution>Lierni Ferna ́ ndez-Ibarrondo Program in Cancer Research IMIM</institution>
        </aff>
        <aff id="aff5">
          <label>5</label>
          <institution>Marta Rodr ́ıguez-Arias Universitat de Valencia</institution>
        </aff>
        <aff id="aff6">
          <label>6</label>
          <institution>Universitat de Valencia</institution>
        </aff>
      </contrib-group>
      <abstract>
        <p />
      </abstract>
    </article-meta>
  </front>
  <body>
    <sec id="sec-1">
      <title>Vincent Warnault</title>
      <p>GReNeC
IMIM - UPF
A`lex Bravo</p>
      <p>GRIB</p>
      <p>IMIM - UPF</p>
    </sec>
    <sec id="sec-2">
      <title>Antonia Serrano</title>
      <p>Instituto de Investigacio´ n
Biome´dica de Ma´laga</p>
    </sec>
    <sec id="sec-3">
      <title>Ferran Sanz</title>
      <p>GRIB
IMIM - UPF</p>
    </sec>
    <sec id="sec-4">
      <title>Marta Portero</title>
      <p>GReNeC
IMIM - UPF</p>
    </sec>
    <sec id="sec-5">
      <title>M.Carmen Blanco-Gand´ıa</title>
      <p>Universitat de Valencia</p>
    </sec>
    <sec id="sec-6">
      <title>Francina Fonseca</title>
      <p>Parc de Salut Mar</p>
      <p>UAB</p>
    </sec>
    <sec id="sec-7">
      <title>Anna Mane´</title>
      <p>Parc de Salut Mar</p>
      <p>UAB</p>
    </sec>
    <sec id="sec-8">
      <title>Roser Nadal</title>
      <p>Institut de Neurocie`ncies</p>
      <p>UAB</p>
    </sec>
    <sec id="sec-9">
      <title>Ezequiel Perez</title>
      <p>Parc de Salut Mar</p>
      <p>UAB</p>
    </sec>
    <sec id="sec-10">
      <title>Marta Torrens</title>
      <p>Parc de Salut Mar</p>
      <p>UAB</p>
    </sec>
    <sec id="sec-11">
      <title>Laura I. Furlong</title>
      <p>GRIB
IMIM - UPF
During the last years there has been a
growing research in the genetics of
psychiatric diseases. However, there is
still a limited understanding of the
cellular and molecular mechanisms leading
to these diseases, which has hampered
the application of this wealth of
knowledge into the clinical practice to
improve diagnosis and treatment of
psychiatric patients. PsyGeNET (http://
www.psygenet.org/) has been
developed to improve the understanding of
psychiatric diseases, by facilitating the
access to the vast amount of their genetic
information in a structured manner,
providing a set of analysis and
visualization tools. In this communication we
describe the protocol we put in place for the
sustainable update of this knowledge
resource. It includes the recruitment of a
team of experts to perform the curation of
the data previously extracted by text
mining. Annotation guidelines and a
webbased annotation tool were developed to
support curators tasks. A curation
workflow was designed including a pilot phase,
and two rounds of curation and
analysis phases. We report the results of the
application of this workflow to the task
of curation of gene-disease associations
for PsyGeNET, including the analysis of
the inter-annotator agreement, and suggest
that this model is a suitable approach for
the sustainable development and update of
knowledge resources.</p>
    </sec>
    <sec id="sec-12">
      <title>1 Introduction</title>
      <p>
        Psychiatric disorders have a great impact on
morbidity and mortality
        <xref ref-type="bibr" rid="ref13 ref7 ref8">(Murray and Lopez, 2013;
Whiteford et al., 2013)</xref>
        . According to the World
Health Organization (WHO), one of every four
people will suffer mental or neurological disorders
        <xref ref-type="bibr" rid="ref1 ref14 ref6">(Kessler et al., 2005; Baldacchino et al., 2009)</xref>
        . It
has been suggested that most psychiatric disorders
display a strong genetic component
        <xref ref-type="bibr" rid="ref11 ref12">(Sullivan et
al., 2012)</xref>
        . During the last years there has been
a growing research in the genetics of psychiatric
disorders, and its findings have been reported on
hundreds of thousands of publications. This
literature constitutes a rich and diverse source of
information essential for any psychiatric research line.
However, the huge amount and continuous growth
of the number of publications refrain scientists to
efficiently explore such large volume of data.
      </p>
      <p>
        PsyGeNET (Psychiatric disorders Gene
association NETwork)
        <xref ref-type="bibr" rid="ref10 ref2 ref5">(Gutie´rrez-Sacrista´n et al., 2015)</xref>
        has been developed to establish a curated resource
on psychiatric diseases and their associated genes.
PsyGeNET integrates knowledge extracted from
the scientific literature by text-mining which has
been curated by experts in psychiatry and
neurosciences.
      </p>
      <p>
        In this communication we describe the process
put in place for the update of the PsyGeNET
database. This involved i) the recruitment of a
team of experts to curate the information extracted
by text-mining; ii) the extraction of information of
gene-disease associations (GDAs) from the
literature using the text mining system BeFree
        <xref ref-type="bibr" rid="ref10 ref2 ref5">(Bravo
et al., 2015)</xref>
        , iii) the development of a curation
workflow (Figure 1), iv) the development of a
web-based annotation tool in order to facilitate
the curation task and v) the definition of detailed
guidelines to assist the curation task.
      </p>
      <p>In particular, we present the results of the Pilot
phase and Curation I phase of the workflow,
including the analysis of the inter-annotator
agreement, and suggest that this protocol is a suitable
approach for the sustainable development and
update of knowledge resources.
A team of 22 experts from different domains (such
as psychiatry, neuroscience, medicine, psychology
and biology) was recruited from the Spanish
Network of Addiction and other collaborators of the
coordination team (Research Group on Integrative
Biomedical Informatics (GRIB)) to participate in
the curation process. The incentives for
participation were to be part of the PsyGeNET team and
to be co-authors in the publication(s) originated
from the project. The curators were trained
during an initial session where the PsyGeNET
annotation guidelines were presented and then during
the Pilot phase. Communication with the
coordination team through e-mail was established to
resolve questions during the curation process. In
addition, on-line and f2f meeting were organized
after key points of the curation process (analysis
phases) to share experiences among all curators
and solve curation issues.
In PsyGeNET, the psychiatric diseases are
identified by UMLS Metathesaurus concepts. Three
experts reviewed the terminology included in more
than 2,000 UMLS concepts related to the
psychiatric disorders of interest, and assigned them to the
following psychiatric disease categories (DCs): 1)
Depressive disorders, 2) Bipolar disorders and
related disorders, 3) Substance/drug induced
depressive disorder, 4) Schizophrenia spectrum and other
psychotic disorders, 5) Drug-induced psychosis,
6) Alcohol use disorders, 7) Cannabis use
disorders and 8) Cocaine use disorders. This
information was used both for text mining of gene-disease
associations by BeFree (see below) and for
identification of disease classes during the curation.
2.3</p>
      <sec id="sec-12-1">
        <title>Text mining of gene-disease associations</title>
        <p>
          BeFree, a text-mining tool that exploits
morphosyntactic information from the text to identify
relationships between biomedical concepts, was
used to identify associations between genes and
the psychiatric diseases of interest from a corpus
of 1M of MEDLINE abstracts focused on
human genetic diseases. The diseases were identified
using the UMLS concepts that define each
disorder, whereas an in-house developed gene
dictionary was used to identify the genes, as described
in
          <xref ref-type="bibr" rid="ref10 ref2 ref5">(Bravo et al., 2015)</xref>
          . The identified disorders
were grouped according to the eight psychiatric
disease categories (described in section 2.2). As
a result, BeFree identified 6,349 associations
between genes and DCs (gene-disease category
association or GDCA) supported by 4,065
publications. A subset of the associations was initially
evaluated by our group to identify the most
frequent text mining errors. For instance, the word
depression is often used in other context in
addition to psychiatry. This initial evaluation was
performed to identify this kind of errors and improve
the text mining system before the identification of
GDCAs. We then applied a number of filters to
reduce the size of the curation task and make it
feasible with the resources at hand. For instance, we
removed associations already present in curated
resources (DisGeNET
          <xref ref-type="bibr" rid="ref10 ref2 ref5">(Pin˜ero et al., 2015)</xref>
          and the
previous release of PsyGeNET ), kept only those
associations published recently (after year 2,000)
in journals with Impact Factor greater than 1, and
we did not take into account reviews. After this
process we obtained 2,507 GDCAs, which were
submitted to expert curation.
2.4
        </p>
      </sec>
      <sec id="sec-12-2">
        <title>Annotation Guidelines</title>
        <p>The PsyGeNET annotation guidelines were
developed with the purpose of guiding the manual
curation process. The guidelines included the
definition of a gene-disease association, how it
should be classified according to the level of
evidence, what information should be considered
for the annotation and provided real examples of
the association types. Finally, it also included a
tutorial on how to use the PsyGeNET annotation
tool. The goal of the curation was to validate the
association of a gene to a particular disease. We
consider that a gene is associated to a disease if
the gene or the product of the gene plays a role
in the disease pathogenesis, or is a marker for the
disease. The PsyGeNET annotation tool was used
to help in this curation task. For each gene-disease
association identified by text mining, the
annotation tool displayed the evidence that supports the
association, more concretely the abstracts and the
sentences in which the gene-disease association
is stated. Then, by inspecting the evidence, the
curator had to determine the type of association
(Association, No Association, False, Error and
Not Clear). The types of association are described
as follows: i) Association: the publication clearly
states that there is an association between the gene
and the disease - it can be a causative association
(e.g. a mutation in the gene causes the disease),
or a marker association (e.g. a SNP in the gene
identified in a GWAS study); ii) No Association:
the publications clearly states that there is no
association between the gene and the disease (e.g.
a publication that reports a negative finding on
the association between the gene and the disease),
iii) False Association: The gene and the disease
are found co-occurring in a sentence, but there is
no clear evidence from the publication that the
gene plays a role or is a marker of the disease and
iv) Error: when there is a text mining error in
the correct identification of the gene and/or the
disease.</p>
        <p>Table 1 shows some examples of the
association types considered in PsyGeNET. In the
example for False Association, the study is on
children that do not meet the criteria for the disease
(FASD) therefore the association between the gene
and the disease has to be classified as false. In
the example of Error, note that in this abstract
OCT is not a gene but an acronym of optical
coherence tomography (OCT). The document
describing the guidelines is available on the
PsyGeNET web page (http://www.psygenet.
org/Psytool_manual_v5.0.pdf). Here
we provide the general instructions for the
curation of the gene-disease associations in
PsyGeNET:
1. The curation has to be performed at abstract
267012 The D-amino acid oxidase
activator gene (G72) has been
found associated with several
psychiatric disorders such as
schizophrenia, major
depression, and bipolar disorder.</p>
        <p>No Association 17692928 There was no association
between TPH-2 gene variants and
MD in the same population that
had shown a strong association
with TPH-1.</p>
        <p>False 25225167 Two children referred for
suspicion of FASD (neither of
which were exposed to alcohol
or met the criteria for FASD)
had a pathogenic
microstructural chromosomal
rearrangement (del16p11.2 of 542 KB
and dup1q44 of 915 KB).</p>
        <p>Error 21174530 OCT demonstrated loss of
foveal depression with
distortion of the foveal architecture in
the macula in all patients
level. For those cases in which abstract is not
clear enough, the full text article should be
reviewed.
2. Annotate only relationships between the gene
and disease. Other types of relationships
should not be annotated.
3. Annotate relationships according to the
provided categories: association, no association,
error, and false.</p>
      </sec>
      <sec id="sec-12-3">
        <title>2.5 Annotation tool</title>
        <p>A user-friendly web-based tool was developed to
assist both the definition of the psychiatric
disorders of interest and curation of gene-disease
associations. The tool was designed to support a
multi-user environment by user and password
assignment. Figure 2 shows a screenshot of the tool
for the curation of GDCAs. The tool shows the
GDCA to be evaluated (in this example the
association between the ETNPPL gene and Bipolar
disorders class), and a publication at a time. The
curator has to review the publication and decide if
the association of the gene and the disease class
holds, and decide on the association type using
the drop-down menu. To aid the curators task, the
tool displays the terminology for the gene
according to standard resources (NCBI Gene, UniProt
and HGNC), and highlights the sentences in which
BeFree identified an association between the gene
and the disease under consideration. If required,
the curator can access the full text article using the
PubMed hyperlink. The curator is also asked to
select a sentence that best represents their validation
decision, if available. This was implemented in
order to collect example sentences to improve the
performance of the BeFree system. In addition,
the tool also provides a progress bar indicating the
number of validations and associations performed
by the expert, and allows to review previous
annotations. We refer to a validation to each
publication supporting a particular GDCA. Note that each
publication can have more than one GDCA.</p>
      </sec>
      <sec id="sec-12-4">
        <title>2.6 Curation workflow</title>
        <p>We put in place a curation workflow including a
pilot phase and two curation and analysis phases
(see Figure 1). During the pilot phase, the initial
training of the curators was carried out including
how to use the curation tool. A set of 100 abstracts
was validated and analyzed during the pilot phase.
After this process both the curation tool and the
annotation guidelines were improved and the first
curation phase was launched (Curation Phase I),
to evaluate 2,507 GDCAs identified by text
mining and supported by 4,065 publications. The
results of the curation were analyzed to estimate the
inter-annotator agreement at the level of abstract.
The validations for which an agreement was not
found in Curation Phase I are then reviewed by a
third expert during Curation Phase II (results not
reported here). Four experts are participating in
this phase. Only the validations for which
agreement of at least 2 experts is found will be included
in the database.
3</p>
      </sec>
    </sec>
    <sec id="sec-13">
      <title>Results and discussion</title>
      <p>Firstly, three experts reviewed the terminology of
2,523 UMLS concepts related to psychiatric
disorders of interest. As a result, 1,942 UMLS
concepts were assigned to one of the 8 disease
categories, being alcohol use disorder, depression
and schizophrenia defined by more than 300
concepts (321, 368 and 488, respectively). On the
other hand, 581 UMLS concepts were excluded
at this stage. Then, BeFree was used to
identify gene-disease associations from the literature
based on the above disease definition and a
subset of the associations focused on the disorders
of interest was selected (see methods section 2.3).
The 2,507 genes associated to DCs identified by
BeFree were submitted to expert curation. These
genes were unevenly distributed across the disease
categories, being schizophrenia the disease
category with more associations followed by
depression and alcohol use disorders (see Figure 3).</p>
      <p>Of note, most of the GDCAs were supported by
only one publication (70.6 %). We included up to
the 5 most recent publications for each GDCA for
the validation process. This led to 242-284
GDCAs to be validated by each curator, depending
on the disease category. Since most of GDCAs
are supported by only one publication, the
number of publications to be reviewed by the
curators ranged between 322 and 491. Before
starting the curation of the 2,507 GDCAs, a Pilot
curation phase was designed with the purpose of
training the curators, testing the PsyGeNET
annotation tool, and reviewing the PsyGeNET
annotation guidelines. One hundred publications were
reviewed during the Pilot phase, distributed in 10
publications per 2 experts. The average
agreement between the experts pairs in the Pilot Phase
was 60%. The main sources of discrepancies were
the handling of speculations, the proper
identification of text mining errors, in particular for genes,
and the distinction between False and Error
Association types. The annotation tool was
modified to show the terminology of the genes in order
to help the curators to find potential errors in the
identification of genes, and by improving the
Review function. Then, the proper curation (Curation
Phase I in the workflow in Figure 1) was launched
and it was completed in 33 days. During Curation
Phase I, 2,507 GDCAs supported by 4,065
publications were reviewed by the curators. Each expert
was assigned with a set of approx. 275 GDCAs
(corresponding to 450 publications) according to
their field of expertise (e.g. Major depression vs
Schizophrenia). Some curators evaluated
associations from all the disease categories, while others
focused in a single category. The results of the
curation phase I were analyzed to identify
agreements and disagreements between the experts.
Table 2 shows the number of abstracts validated by
each curator team (composed of two experts) and
the agreement achieved. The average agreement
between all the experts was 68.95%, higher that
the one obtained in the Pilot Phase. For one
curator team the agreement was higher (89%) than for
the rest of the teams. We can attribute this higher
agreement to the fact that there was some
communication between the two experts to discuss on the
curation criteria during the Curation Phase I.</p>
      <p>From the validations in which agreement was
found (2,813 validations), 1,880 were classified as
Association or No Association; 901 were
classified as False or Error, and only in 32 of them, the
evidence extracted from the publication was not
enough to classify them within any of the previous
categories, falling in the not clear category
(Figure 4). The set of 1,880 validations will be part
of the next release of PsyGeNET. Notably, an
important fraction of these associations (24.7%) are
classified as No association, meaning that there is
evidence reporting negative findings on the
association between the gene and the disease. This
highlights the importance of recording negative
findings from the literature in knowledge resources.
On the other hand, collecting these information is
relevant for the development of corpora for
training text mining systems able to identify negative
findings regarding gene-disease associations from
the literature.</p>
      <p>We observe that for 30% of the total GDCAs
validated, agreement between curators was not
found. A substantial fraction of the disagreements
involved the annotation of an association as False
by one of the experts (53.28%, see Figure 5).
The results of Curation Phase I were discussed
with the experts in order to identify the main
difficulties during the annotation. The main
sources of the discrepancies between curators
were the following: i) difficulty in assessing if
the studies using animal models captures well the
disease pathophysiology, ii) the studies focused
on pharmacogenomics or response to drug
treatments, iii) studies assessing disease phenotypes
(e.g. low mood) in otherwise normal populations,
and iv) the assessment of validity of the statistical
analysis in some studies (e.g. GWAS studies).
In the first case, the decision on the association
type will depend on the expertise of the curator
on animal model research in psychiatry, that was
not the same among the team of experts. In the
other three cases the experts expressed difficulties
in correctly identifying if an association has to be
annotated or not. Overall, although the curation
task was very focused to the domain of genetics
of psychiatric diseases, the wide variety of studies
covered by the publications (GWAs studies,
sequencing studies, animal models, etc) require
an equivalent diversity of expertise among the
experts. We think that this complexity in the task
is one of the main reasons for the inter-annotator
agreement achieved. Ongoing work includes
revisiting the annotation guidelines to further
clarify the curation issues raised, in order to
improve the agreement in the annotations.</p>
      <p>
        In recent years, many efforts have been made
to develop and contribute with novel corpora
in the biomedical domain. Nevertheless, the
number of corpora annotated with information
on gene-disease associations is particularly low
        <xref ref-type="bibr" rid="ref9">(Neves, 2014)</xref>
        . For example, the Craven
corpus
        <xref ref-type="bibr" rid="ref4">(Craven et al., 1999)</xref>
        , contains annotations
of gene-disease associations, but there is no
information on data quality such as inter-annotator
agreement in the original publication. The
EUADR corpus
        <xref ref-type="bibr" rid="ref11 ref12">(Van Mulligen et al., 2012)</xref>
        includes associations between genes and diseases
from 100 MEDLINE abstracts, with an
interannotator agreement of 86%. Wiegers et al.
presented the manual curation of
chemical-genedisease network for the Comparative
Toxicogenomics Database (CTD)
        <xref ref-type="bibr" rid="ref1 ref14">(Wiegers et al., 2009)</xref>
        .
For this study 112 articles were distributed
between three curators (each one revised less than
60 articles), achieving an inter-annotator
agreement of 77%. The CoMAGC corpus
        <xref ref-type="bibr" rid="ref13 ref7">(Lee et al.,
2013)</xref>
        , focused on genes associated to prostate,
breast and ovarian cancer, is based on 821
sentences. The authors report an agreement 72%. In
another study, agreement over 70% was reported
in the development of a sentence-based corpus on
prostate cancer-gene associations
        <xref ref-type="bibr" rid="ref3">(Chun et al.,
2006)</xref>
        . In summary, compared to other corpora
annotation initiatives, our inter-annotator agreement
results are lower. As described in the paragraphs
above, we think that the agreement obtained is due
to the complexity of the annotation task. In
addition, the large number of experts (for instance, 22
in our case vs 5 in the case of the EU-ADR
corpus) and also the large size of our corpus (4,065
publications vs approx. 100 in EU-ADR and CTD
corpora) could also explain the lower agreement
obtained compared to other curation initiatives.
      </p>
      <p>The Curation Phase II is aimed at reviewing
the associations in which no agreement was found
among two experts in the first phase of curation.
Currently, this involves 1,252 validations, which
are being reviewed by a third expert (ongoing
work at the time of writing). Finally, the
information that will be included in PsyGeNET are the
associations in which at least two experts agreed
on the annotation.
4</p>
    </sec>
    <sec id="sec-14">
      <title>Conclusions</title>
      <p>In this communication we report the development
of a protocol for the sustainable update of a
knowledge resource on the genetics of psychiatric
diseases, PysGeNET. We combined state-of-the-art
text-mining, data filtering and curation by a
community of domain experts for the release of a new
version of the database. We designed a
protocol that includes curators’ training and the
iterative improvement of both the tools and
annotation guidelines. The proposed approach is
allowing to update the database in a timely manner with
expert-validated information. Importantly, our
curation protocol included the identification of
negative findings from the literature. Note that 24.7%
of the GDCAs were classified as No association,
indicating the importance of properly annotating
this information in a knowledge resource. This
information will be taken into account for the
ranking of the gene-disease association in the next
release of PsyGeNET. In addition, the corpus of
annotated sentences and abstracts developed during
the curation constitutes a valuable resource for the
development and evaluation of relation extraction
systems. In this era of biomedical big data, we
present this approach involving the expert
community for the curation of the information as a
suitable approach for the development and
maintenance of knowledge resources.
5</p>
    </sec>
    <sec id="sec-15">
      <title>Fundings</title>
      <p>We received support from ISCIII-FEDER
(PI13/00082, CP10/00524), IMI-JU under grants
agreements n 115002 (eTOX), n 115191 (Open
PHACTS)], n 115372 (EMIF) and n 115735
(iPiE), resources of which are composed of
nancial contribution from the EU-FP7
(FP7/20072013) and EFPIA companies in kind contribution,
and the EU H2020 Programme 2014-2020 under
grant agreements no. 634143
(MedBioinformatics) and no. 676559 (Elixir-Excelerate). The
Research Programme on Biomedical Informatics
(GRIB) is a node of the Spanish National Institute
of Bioinformatics (INB).</p>
    </sec>
  </body>
  <back>
    <ref-list>
      <ref id="ref1">
        <mixed-citation>
          [Baldacchino et al.
          <year>2009</year>
          ]
          <string-name>
            <given-names>A</given-names>
            <surname>Baldacchino</surname>
            , N GroussardEscaffre
          </string-name>
          ,
          <string-name>
            <given-names>C</given-names>
            <surname>Clancy</surname>
          </string-name>
          ,
          <string-name>
            <given-names>C</given-names>
            <surname>Lack</surname>
          </string-name>
          ,
          <string-name>
            <given-names>K</given-names>
            <surname>Sieroslavrska</surname>
          </string-name>
          ,
          <string-name>
            <surname>C-L Hodges</surname>
          </string-name>
          ,
          <string-name>
            <surname>L-B Merinder</surname>
            ,
            <given-names>T</given-names>
          </string-name>
          <string-name>
            <surname>Greacen</surname>
            ,
            <given-names>M</given-names>
          </string-name>
          <string-name>
            <surname>Sorsa</surname>
            ,
            <given-names>H</given-names>
          </string-name>
          <string-name>
            <surname>Laijarvi</surname>
          </string-name>
          , et al.
          <year>2009</year>
          .
          <article-title>Epidemiological issues in comorbidity: lessons learnt from a pan-european isadora project</article-title>
          .
          <source>Mental Health and Substance Use: Dual Diagnosis</source>
          ,
          <volume>2</volume>
          (
          <issue>2</issue>
          ):
          <fpage>88</fpage>
          -
          <lpage>100</lpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref2">
        <mixed-citation>
          [Bravo et al.
          <year>2015</year>
          ]
          <article-title>A` lex Bravo, Janet Pin˜ero, Nu´ria Queralt-Rosinach, Michael Rautschka</article-title>
          , and
          <string-name>
            <surname>Laura</surname>
            <given-names>I</given-names>
          </string-name>
          <string-name>
            <surname>Furlong</surname>
          </string-name>
          .
          <year>2015</year>
          .
          <article-title>Extraction of relations between genes and diseases from text and large-scale data analysis: implications for translational research</article-title>
          .
          <source>BMC bioinformatics</source>
          ,
          <volume>16</volume>
          (
          <issue>1</issue>
          ):
          <fpage>1</fpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref3">
        <mixed-citation>
          [Chun et al.2006]
          <string-name>
            <surname>Hong-Woo</surname>
            <given-names>Chun</given-names>
          </string-name>
          , Yoshimasa Tsuruoka,
          <string-name>
            <surname>Jin-Dong</surname>
            <given-names>Kim</given-names>
          </string-name>
          , Rie Shiba, Naoki Nagata, Teruyoshi Hishiki, and
          <string-name>
            <surname>Jun'ichi Tsujii</surname>
          </string-name>
          .
          <year>2006</year>
          .
          <article-title>Automatic recognition of topic-classified relations between prostate cancer and genes using medline abstracts</article-title>
          .
          <source>BMC bioinformatics</source>
          ,
          <volume>7</volume>
          (
          <issue>3</issue>
          ):
          <fpage>1</fpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref4">
        <mixed-citation>
          [Craven et al.1999]
          <string-name>
            <given-names>Mark</given-names>
            <surname>Craven</surname>
          </string-name>
          ,
          <string-name>
            <given-names>Johan</given-names>
            <surname>Kumlien</surname>
          </string-name>
          , et al.
          <year>1999</year>
          .
          <article-title>Constructing biological knowledge bases by extracting information from text sources</article-title>
          .
          <source>In ISMB</source>
          , volume
          <year>1999</year>
          , pages
          <fpage>77</fpage>
          -
          <lpage>86</lpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref5">
        <mixed-citation>
          <article-title>[Gutie´rrez-</article-title>
          <string-name>
            <surname>Sacrista´</surname>
          </string-name>
          n et al.2015]
          <article-title>Alba Gutie´rrezSacrista´n</article-title>
          , Sole`ne Grosdidier, Olga Valverde,
          <string-name>
            <given-names>Marta</given-names>
            <surname>Torrens</surname>
          </string-name>
          ,
          <string-name>
            <surname>A</surname>
          </string-name>
          ` lex Bravo, Janet Pin˜ero,
          <source>Ferran Sanz, and Laura I Furlong</source>
          .
          <year>2015</year>
          .
          <article-title>Psygenet: a knowledge platform on psychiatric disorders and their genes</article-title>
          .
          <source>Bioinformatics</source>
          , page btv301.
        </mixed-citation>
      </ref>
      <ref id="ref6">
        <mixed-citation>
          [Kessler et al.2005
          <string-name>
            <surname>] Ronald</surname>
            <given-names>C Kessler</given-names>
          </string-name>
          , Patricia Berglund, Olga Demler, Robert Jin,
          <string-name>
            <surname>Kathleen R Merikangas</surname>
            , and
            <given-names>Ellen E</given-names>
          </string-name>
          <string-name>
            <surname>Walters</surname>
          </string-name>
          .
          <year>2005</year>
          .
          <article-title>Lifetime prevalence and age-of-onset distributions of dsm-iv disorders in the national comorbidity survey replication</article-title>
          .
          <source>Archives of general psychiatry</source>
          ,
          <volume>62</volume>
          (
          <issue>6</issue>
          ):
          <fpage>593</fpage>
          -
          <lpage>602</lpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref7">
        <mixed-citation>
          <string-name>
            <surname>[Lee</surname>
          </string-name>
          et al.2013]
          <string-name>
            <surname>Hee-Jin</surname>
            <given-names>Lee</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Sang-Hyung</surname>
            <given-names>Shim</given-names>
          </string-name>
          , MiRyoung Song,
          <string-name>
            <given-names>Hyunju</given-names>
            <surname>Lee</surname>
          </string-name>
          , and Jong C Park.
          <year>2013</year>
          .
          <article-title>Comagc: a corpus with multi-faceted annotations of gene-cancer relations</article-title>
          .
          <source>BMC bioinformatics</source>
          ,
          <volume>14</volume>
          (
          <issue>1</issue>
          ):
          <fpage>1</fpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref8">
        <mixed-citation>
          <article-title>[Murray and Lopez2013] Christopher JL Murray</article-title>
          and
          <string-name>
            <given-names>Alan D</given-names>
            <surname>Lopez</surname>
          </string-name>
          .
          <year>2013</year>
          .
          <article-title>Measuring the global burden of disease</article-title>
          .
          <source>New England Journal of Medicine</source>
          ,
          <volume>369</volume>
          (
          <issue>5</issue>
          ):
          <fpage>448</fpage>
          -
          <lpage>457</lpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref9">
        <mixed-citation>
          [Neves2014]
          <string-name>
            <given-names>Mariana</given-names>
            <surname>Neves</surname>
          </string-name>
          .
          <year>2014</year>
          .
          <article-title>An analysis on the entity annotations in biological corpora</article-title>
          .
          <source>F1000Research</source>
          , 3.
        </mixed-citation>
      </ref>
      <ref id="ref10">
        <mixed-citation>
          [Pin˜ero et al.2015]
          <article-title>Janet Pin˜ero, Nu´ria QueraltRosinach, A`lex Bravo, Jordi Deu-Pons, Anna Bauer-Mehren, Martin Baron</article-title>
          ,
          <source>Ferran Sanz, and Laura I Furlong</source>
          .
          <year>2015</year>
          .
          <article-title>Disgenet: a discovery platform for the dynamical exploration of human diseases and their genes</article-title>
          .
          <source>Database</source>
          ,
          <year>2015</year>
          :
          <fpage>bav028</fpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref11">
        <mixed-citation>
          [Sullivan et al.2012
          <string-name>
            <surname>] Patrick F Sullivan</surname>
          </string-name>
          ,
          <string-name>
            <surname>Mark J Daly</surname>
          </string-name>
          , and
          <string-name>
            <surname>Michael O'Donovan</surname>
          </string-name>
          .
          <year>2012</year>
          .
          <article-title>Genetic architectures of psychiatric disorders: the emerging picture and its implications</article-title>
          .
          <source>Nature Reviews Genetics</source>
          ,
          <volume>13</volume>
          (
          <issue>8</issue>
          ):
          <fpage>537</fpage>
          -
          <lpage>551</lpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref12">
        <mixed-citation>
          <string-name>
            <surname>[Van</surname>
          </string-name>
          Mulligen et al.2012
          <string-name>
            <surname>] Erik M Van Mulligen</surname>
          </string-name>
          ,
          <string-name>
            <surname>Annie</surname>
            Fourrier-Reglat, David Gurwitz,
            <given-names>Mariam</given-names>
          </string-name>
          <string-name>
            <surname>Molokhia</surname>
          </string-name>
          , Ainhoa Nieto, Gianluca Trifiro,
          <article-title>Jan A Kors,</article-title>
          and
          <string-name>
            <surname>Laura</surname>
            <given-names>I</given-names>
          </string-name>
          <string-name>
            <surname>Furlong</surname>
          </string-name>
          .
          <year>2012</year>
          .
          <article-title>The eu-adr corpus: annotated drugs, diseases, targets, and their relationships</article-title>
          .
          <source>Journal of biomedical informatics</source>
          ,
          <volume>45</volume>
          (
          <issue>5</issue>
          ):
          <fpage>879</fpage>
          -
          <lpage>884</lpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref13">
        <mixed-citation>
          [Whiteford et al.2013]
          <article-title>Harvey A Whiteford, Louisa Degenhardt</article-title>
          , Ju¨rgen Rehm, Amanda J Baxter, Alize J Ferrari, Holly E Erskine, Fiona J Charlson, Rosana E Norman,
          <string-name>
            <surname>Abraham D Flaxman</surname>
            ,
            <given-names>Nicole</given-names>
          </string-name>
          <string-name>
            <surname>Johns</surname>
          </string-name>
          , et al.
          <year>2013</year>
          .
          <article-title>Global burden of disease attributable to mental and substance use disorders: findings from the global burden of disease study 2010</article-title>
          .
          <source>The Lancet</source>
          ,
          <volume>382</volume>
          (
          <issue>9904</issue>
          ):
          <fpage>1575</fpage>
          -
          <lpage>1586</lpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref14">
        <mixed-citation>
          [Wiegers et al.2009]
          <string-name>
            <surname>Thomas</surname>
            <given-names>C Wiegers</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Allan P Davis</surname>
            ,
            <given-names>K Bretonnel</given-names>
          </string-name>
          <string-name>
            <surname>Cohen</surname>
          </string-name>
          , Lynette Hirschman, and
          <string-name>
            <surname>Carolyn</surname>
          </string-name>
          J Mattingly.
          <year>2009</year>
          .
          <article-title>Text mining and manual curation of chemical-gene-disease networks for the comparative toxicogenomics database (ctd)</article-title>
          .
          <source>BMC bioinformatics</source>
          ,
          <volume>10</volume>
          (
          <issue>1</issue>
          ):
          <fpage>326</fpage>
          .
        </mixed-citation>
      </ref>
    </ref-list>
  </back>
</article>