<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.0 20120330//EN" "JATS-archivearticle1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta>
      <journal-title-group>
        <journal-title>and Ca´ssia Tro-
jahn dos Santos. Ontology alignment evaluation initiative: six years of experience. Journal
on Data Semantics</journal-title>
      </journal-title-group>
    </journal-meta>
    <article-meta>
      <title-group>
        <article-title>Results of the Ontology Alignment Evaluation Initiative 2015?</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <string-name>Michelle Cheatham</string-name>
          <email>michelle.cheatham@wright.edu</email>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Zlatan Dragisic</string-name>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Je´roˆ me Euzenat</string-name>
          <email>Jerome.Euzenat@inria.fr</email>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Daniel Faria</string-name>
          <email>dfaria@igc.gulbenkian.pt</email>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Alfio Ferrara</string-name>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Giorgos Flouris</string-name>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Irini Fundulaki</string-name>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Roger Granada</string-name>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Valentina Ivanova</string-name>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ernesto Jime´nez-Ruiz</string-name>
          <email>ernesto@cs.ox.ac.uk</email>
          <xref ref-type="aff" rid="aff11">11</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Patrick Lambrix</string-name>
          <email>patrick.lambrixg@liu.se</email>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Stefano Montanelli</string-name>
          <email>stefano.montanellig@unimi.it</email>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Catia Pesquita</string-name>
          <email>cpesquita@di.fc.ul.pt</email>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Tzanina Saveta</string-name>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Pavel Shvaiko</string-name>
          <email>pavel.shvaiko@infotn.it</email>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Alessandro Solimando</string-name>
          <email>alessandro.solimando@inria.fr</email>
          <xref ref-type="aff" rid="aff2">2</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ca´ssia Trojahn</string-name>
          <email>cassia.trojahng@irit.fr</email>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ondrˇej Zamazal</string-name>
          <email>ondrej.zamazal@vse.cz</email>
          <xref ref-type="aff" rid="aff10">10</xref>
        </contrib>
        <aff id="aff0">
          <label>0</label>
          <institution>Data Semantics (DaSe) Laboratory, Wright State University</institution>
          ,
          <country country="US">USA</country>
        </aff>
        <aff id="aff1">
          <label>1</label>
          <institution>INRIA &amp; Univ. Grenoble Alpes</institution>
          ,
          <addr-line>Grenoble</addr-line>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff2">
          <label>2</label>
          <institution>INRIA-Saclay &amp; Univ. Paris-Sud</institution>
          ,
          <addr-line>Orsay</addr-line>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff3">
          <label>3</label>
          <institution>IRIT &amp; Universite ́ Toulouse II</institution>
          ,
          <addr-line>Toulouse</addr-line>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff4">
          <label>4</label>
          <institution>Institute of Computer Science-FORTH</institution>
          ,
          <addr-line>Heraklion</addr-line>
          ,
          <country country="GR">Greece</country>
        </aff>
        <aff id="aff5">
          <label>5</label>
          <institution>Instituto Gulbenkian de Cieˆncia</institution>
          ,
          <addr-line>Lisbon</addr-line>
          ,
          <country country="PT">Portugal</country>
        </aff>
        <aff id="aff6">
          <label>6</label>
          <institution>LASIGE, Faculdade de Cieˆncias, Universidade de Lisboa</institution>
          ,
          <country country="PT">Portugal</country>
        </aff>
        <aff id="aff7">
          <label>7</label>
          <institution>Linko ̈ping University &amp; Swedish e-Science Research Center</institution>
          ,
          <addr-line>Linko ̈ping</addr-line>
          ,
          <country country="SE">Sweden</country>
        </aff>
        <aff id="aff8">
          <label>8</label>
          <institution>TasLab</institution>
          ,
          <addr-line>Informatica Trentina, Trento</addr-line>
          ,
          <country country="IT">Italy</country>
        </aff>
        <aff id="aff9">
          <label>9</label>
          <institution>Universita` degli studi di Milano</institution>
          ,
          <country country="IT">Italy</country>
        </aff>
        <aff id="aff10">
          <label>10</label>
          <institution>University of Economics</institution>
          ,
          <addr-line>Prague</addr-line>
          ,
          <country country="CZ">Czech Republic</country>
        </aff>
        <aff id="aff11">
          <label>11</label>
          <institution>University of Oxford</institution>
          ,
          <country country="UK">UK</country>
        </aff>
      </contrib-group>
      <pub-date>
        <year>2015</year>
      </pub-date>
      <volume>8797</volume>
      <fpage>73</fpage>
      <lpage>126</lpage>
      <abstract>
        <p>Ontology matching consists of finding correspondences between semantically related entities of two ontologies. OAEI campaigns aim at comparing ontology matching systems on precisely defined test cases. These test cases can use ontologies of different nature (from simple thesauri to expressive OWL ontologies) and use different modalities, e.g., blind evaluation, open evaluation and consensus. OAEI 2015 offered 8 tracks with 15 test cases followed by 22 participants. Since 2011, the campaign has been using a new evaluation modality which provides more automation to the evaluation. This paper is an overall presentation of the OAEI 2015 campaign.</p>
      </abstract>
    </article-meta>
  </front>
  <body>
    <sec id="sec-1">
      <title>Introduction</title>
      <p>The Ontology Alignment Evaluation Initiative1 (OAEI) is a coordinated international
initiative, which organizes the evaluation of the increasing number of ontology
matching systems [14, 17]. The main goal of OAEI is to compare systems and algorithms on
the same basis and to allow anyone for drawing conclusions about the best matching
strategies. Our ambition is that, from such evaluations, tool developers can improve
their systems.</p>
      <p>
        Two first events were organized in 2004: (i) the Information Interpretation and
Integration Conference (I3CON) held at the NIST Performance Metrics for Intelligent
Systems (PerMIS) workshop and (ii) the Ontology Alignment Contest held at the
Evaluation of Ontology-based Tools (EON) workshop of the annual International Semantic
Web Conference (ISWC) [38]. Then, a unique OAEI campaign occurred in 2005 at the
workshop on Integrating Ontologies held in conjunction with the International
Conference on Knowledge Capture (K-Cap) [
        <xref ref-type="bibr" rid="ref2">2</xref>
        ]. Starting from 2006 through 2014 the OAEI
campaigns were held at the Ontology Matching workshops collocated with ISWC [
        <xref ref-type="bibr" rid="ref1 ref4 ref6 ref9">15,
13, 4, 10–12, 1, 6, 9</xref>
        ]. In 2015, the OAEI results were presented again at the Ontology
Matching workshop2 collocated with ISWC, in Bethlehem, PA US.
      </p>
      <p>Since 2011, we have been using an environment for automatically processing
evaluations (x2.2), which has been developed within the SEALS (Semantic Evaluation At
Large Scale) project3. SEALS provided a software infrastructure, for automatically
executing evaluations, and evaluation campaigns for typical semantic web tools, including
ontology matching. For OAEI 2015, almost all of the OAEI data sets were evaluated
under the SEALS modality, providing a more uniform evaluation setting. This year we
did not continue the library track, however we significantly extended the evaluation
concerning the conference, interactive and instance matching tracks. Furthermore, the
multifarm track was extended with Arabic and Italian as languages.</p>
      <p>This paper synthetizes the 2015 evaluation campaign and introduces the results
provided in the papers of the participants. The remainder of the paper is organised as
follows. In Section 2, we present the overall evaluation methodology that has been used.
Sections 3-9 discuss the settings and the results of each of the test cases. Section 11
overviews lessons learned from the campaign. Finally, Section 12 concludes the paper.
2</p>
    </sec>
    <sec id="sec-2">
      <title>General methodology</title>
      <p>We first present the test cases proposed this year to the OAEI participants (x2.1). Then,
we discuss the resources used by participants to test their systems and the execution
environment used for running the tools (x2.2). Next, we describe the steps of the OAEI
campaign (x2.3-2.5) and report on the general execution of the campaign (x2.6).</p>
      <sec id="sec-2-1">
        <title>1 http://oaei.ontologymatching.org 2 http://om2015.ontologymatching.org 3 http://www.seals-project.eu</title>
        <p>2.1</p>
        <sec id="sec-2-1-1">
          <title>Tracks and test cases</title>
          <p>This year’s campaign consisted of 8 tracks gathering 15 test cases and different
evaluation modalities:
The benchmark track (x3): Like in previous campaigns, a systematic benchmark
series has been proposed. The goal of this benchmark series is to identify the areas
in which each matching algorithm is strong or weak by systematically altering an
ontology. This year, we generated a new benchmark based on the original
bibliographic ontology and another benchmark using an energy ontology.</p>
          <p>The expressive ontology track offers real world ontologies using OWL modelling
capabilities:
Anatomy (x4): The anatomy test case is about matching the Adult Mouse
Anatomy (2744 classes) and a small fragment of the NCI Thesaurus (3304
classes) describing the human anatomy.</p>
          <p>Conference (x5): The goal of the conference test case is to find all correct
correspondences within a collection of ontologies describing the domain of
organizing conferences. Results were evaluated automatically against reference
alignments and by using logical reasoning techniques.</p>
          <p>Large biomedical ontologies (x6): The largebio test case aims at finding
alignments between large and semantically rich biomedical ontologies such as
FMA, SNOMED-CT, and NCI. The UMLS Metathesaurus has been used as
the basis for reference alignments.</p>
        </sec>
        <sec id="sec-2-1-2">
          <title>Multilingual</title>
          <p>Multifarm (x7): This test case is based on a subset of the Conference data set,
translated into eight different languages (Chinese, Czech, Dutch, French,
German, Portuguese, Russian, and Spanish) and the corresponding alignments
between these ontologies. Results are evaluated against these alignments. This
year, translations involving Arabic and Italian languages have been added.</p>
        </sec>
        <sec id="sec-2-1-3">
          <title>Interactive matching</title>
          <p>Interactive (x8): This test case offers the possibility to compare different
matching tools which can benefit from user interaction. Its goal is to show if user
interaction can improve matching results, which methods are most promising
and how many interactions are necessary. Participating systems are evaluated
on the conference data set using an oracle based on the reference alignment.
Ontology Alignment For Query Answering OA4QA (x9): This test case offers the
possibility to evaluate alignments in their ability to enable query answering
in an ontology based data access scenario, where multiple aligned ontologies
exist. In addition, the track is intended as a possibility to study the practical
effects of logical violations affecting the alignments, and to compare the
different repair strategies adopted by the ontology matching systems. In order to
facilitate the understanding of the dataset and the queries, the conference data
set is used, extended with synthetic ABoxes.</p>
          <p>Instance matching (x10). The track is organized in five independent tasks and each
task is articulated in two tests, namely sandbox and mainbox, with different scales,
i.e., number of instances to match. The sandbox (small scale) is an open test,
meaning that the set of expected mappings (i.e., reference alignment) is given in advance
test formalism relations confidence modalities
language
benchmark</p>
          <p>anatomy
conference</p>
          <p>largebio
multifarm
interactive</p>
          <p>OA4QA
author-dis
author-rec</p>
          <p>val-sem
val-struct
val-struct-sem</p>
          <p>OWL
OWL
OWL
OWL
OWL
OWL
OWL
OWL
OWL
OWL
OWL
OWL
=
=
=, &lt;=
=
=
=, &lt;=
=, &lt;=
=
=
&lt;=
&lt;=
&lt;=</p>
          <p>to the participants. The mainbox (medium scale) is a blind test, meaning that the
reference alignment is not given in advance to the participants. Each test contains
two datasets called source and target and the goal is to discover the matching pairs,
i.e., mappings or correspondences, among the instances in the source dataset and
those in the target dataset.</p>
          <p>Author-dis: The goal of the author-dis task is to link OWL instances referring to
the same person (i.e., author) based on their publications.</p>
          <p>Author-rec: The goal of the author-rec task is to associate a person, i.e., author,
with the corresponding publication report containing aggregated information
about the publication activity of the person, such as number of publications,
h-index, years of activity, number of citations.</p>
          <p>Val-sem: The goal of the val-sem task is to determine when two OWL instances
describe the same Creative Work. The datasets of the val-sem task have been
produced by altering a set of original data through value-based and
semanticsaware transformations.</p>
          <p>Val-struct: The goal of the val-struct task is to determine when two OWL
instances describe the same Creative Work. The datasets of the val-struct task
have been produced by altering a set of original data through value-based and
structure-based transformations.</p>
          <p>Val-struct-sem: The goal of the val-struct-sem task is to determine when two
OWL instances describe the same Creative Work. The datasets of the
val-structsem task have been produced by altering a set of original data through
valuebased, structure-based and semantics-aware transformations.
Since 2011, tool developers had to implement a simple interface and to wrap their tools
in a predefined way including all required libraries and resources. A tutorial for tool
wrapping was provided to the participants. It describes how to wrap a tool and how to
use a simple client to run a full evaluation locally. After local tests are passed
successfully, the wrapped tool has to be uploaded on the SEALS portal4. Consequently, the
evaluation can be executed by the organizers with the help of the SEALS
infrastructure. This approach allowed to measure runtime and ensured the reproducibility of the
results. As a side effect, this approach also ensures that a tool is executed with the same
settings for all of the test cases that were executed in the SEALS mode.
2.3</p>
        </sec>
        <sec id="sec-2-1-4">
          <title>Preparatory phase</title>
          <p>Ontologies to be matched and (where applicable) reference alignments have been
provided in advance during the period between June 15th and July 3rd, 2015. This gave
potential participants the occasion to send observations, bug corrections, remarks and
other test cases to the organizers. The goal of this preparatory period is to ensure that
the delivered tests make sense to the participants. The final test base was released on
July 3rd, 2015. The (open) data sets did not evolve after that.
2.4</p>
        </sec>
        <sec id="sec-2-1-5">
          <title>Execution phase</title>
          <p>
            During the execution phase, participants used their systems to automatically match the
test case ontologies. In most cases, ontologies are described in OWL-DL and serialized
in the RDF/XML format [
            <xref ref-type="bibr" rid="ref8">8</xref>
            ]. Participants can self-evaluate their results either by
comparing their output with reference alignments or by using the SEALS client to compute
precision and recall. They can tune their systems with respect to the non blind
evaluation as long as the rules published on the OAEI web site are satisfied. This phase has
been conducted between July 3rd and September 1st, 2015.
2.5
          </p>
        </sec>
        <sec id="sec-2-1-6">
          <title>Evaluation phase</title>
          <p>Participants have been encouraged to upload their wrapped tools on the SEALS portal
by September 1st, 2015. For the SEALS modality, a full-fledged test including all
submitted tools has been conducted by the organizers and minor problems were reported
to some tool developers, who had the occasion to fix their tools and resubmit them.</p>
          <p>First results were available by October 1st, 2015. The organizers provided these
results individually to the participants. The results were published on the respective
web pages by the organizers by October 15st. The standard evaluation measures are
usually precision and recall computed against the reference alignments. More details
on evaluation measures are given in each test case section.
4 http://www.seals-project.eu/join-the-community/
2.6</p>
        </sec>
        <sec id="sec-2-1-7">
          <title>Comments on the execution</title>
          <p>The number of participating systems has changed over the years with an increase
tendency with some exceptional cases: 4 participants in 2004, 7 in 2005, 10 in 2006, 17
in 2007, 13 in 2008, 16 in 2009, 15 in 2010, 18 in 2011, 21 in 2012, 23 in 2013, 14 in
2014. This year, we count on 22 systems. Furthermore participating systems are
constantly changing, for example, this year 10 systems had not participated in any of the
previous campaigns. The list of participants is summarized in Table 2. Note that some
systems were also evaluated with different versions and configurations as requested by
developers (see test case sections for details).</p>
          <p>e
it</p>
          <p>D re -L io
LM LANO ANOMM trchoaM -PAOM -PAOM ANO pa +T isOM gpaM gpaCM -gpaBM tgpaLM
A C C C KD KD EX GM IsnM rvJa ilyL L L L L</p>
          <p>o o o o
p p p p p p p p p</p>
          <p>p p p p p p p
p p p p p p p p p p p
p p p p p p p p p p
p p p p p p p p</p>
          <p>p
p p p p p
p p p p p</p>
          <p>p p
Finally, some systems were not able to pass some test cases as indicated in Table 2.</p>
          <p>The result summary per test case is presented in the following sections.
3</p>
        </sec>
      </sec>
    </sec>
    <sec id="sec-3">
      <title>Benchmark</title>
      <p>3.1</p>
      <sec id="sec-3-1">
        <title>Test data</title>
        <p>The goal of the benchmark data set is to provide a stable and detailed picture of each
algorithm. For that purpose, algorithms are run on systematically generated test cases.
The systematic benchmark test set is built around a seed ontology and many variations
of it. Variations are artificially generated by discarding and modifying features from a
seed ontology. Considered features are names of entities, comments, the specialization
hierarchy, instances, properties and classes. This test focuses on the characterization of
the behavior of the tools rather than having them compete on real-life problems. Full
description of the systematic benchmark test set can be found on the OAEI web site.</p>
        <p>Since OAEI 2011.5, the test sets are generated automatically by the test generator
described in [16] from different seed ontologies. This year, we used two ontologies:
biblio The bibliography ontology used in the previous years which concerns
bibliographic references and is inspired freely from BibTeX;
energy energyresource5 is an ontology representing energy information for smart home
systems developed at the Technische Universita¨t Wien.</p>
        <p>The characteristics of these ontologies are described in Table 3.</p>
        <p>Test set</p>
        <p>biblio energy
classes+prop 33+64 523+110
instances 112 16
entities 209 723
triples 1332 9331</p>
        <p>The initially generated tests from the IFC4 ontology which was provided to
participants was found to be “somewhat erroneous” as the reference alignments contained
only entities in the prime ontology namespace. We thus generated the energy data set.
This test has also created problems to some systems, but we decided to keep it as an
example, especially that some other systems have worked on it regularly with decent
results. Hence, it may be useful for developers to understand why this is the case.</p>
        <p>The energy data set was not available to participants when they submitted their
systems. The tests were also blind for the organizers since we did not look into them
before running the systems.</p>
        <p>The reference alignments are still restricted to named classes and properties and use
the “=” relation with confidence of 1.
3.2</p>
      </sec>
      <sec id="sec-3-2">
        <title>Results</title>
        <p>Contrary to previous years, we have not been able to evaluate the systems in a uniform
setting. This is mostly due to relaxing the policy for systems which were not properly
packaged under the SEALS interface so that they could be seamlessly evaluated.
Systems required extra software installation and extra software licenses which rendered
evaluation uneasy.</p>
        <p>Another reason of this situation is the limited availability of evaluators for installing
software for the purpose of evaluation.
5 https://www.auto.tuwien.ac.at/downloads/thinkhome/ontology/
EnergyResourceOntology.owl</p>
        <p>It was actually the goal of the SEALS project to automate this evaluation so that
the tool installation burden was put on tool developers and the evaluation burden on
evaluators. This also reflects the idea that a good tool is a tool easy to install, so in
which the user does not have many reasons to not using it.</p>
        <p>As a consequence, systems have been evaluated in three different machine
configurations:
– edna, AML2014, AML, CroMatcher, GMap, Lily, LogMap-C, LogMapLt, LogMap and
XMap were run on a Debian Linux virtual machine configured with four
processors and 8GB of RAM running under a Dell PowerEdge T610 with 2*Intel Xeon
Quad Core 2.26GHz E5607 processors and 32GB of RAM, under Linux ProxMox
2 (Debian). All matchers where run under the SEALS client using Java 1.8 and a
maximum heap size of 8GB.
– DKP-AOM, JarvisOM, RSDLWB and ServOMBI were run on a Debian Linux virtual
machine configured with four processors and 20GB of RAM running under a Dell
PowerEdge T610 with 2*Intel Xeon Quad Core 2.26GHz E5607 processors and
32GB of RAM, under Linux ProxMox 2 (Debian).
– Mamba was run under Ubuntu 14.04 on a Intel Core i7-3537U 2.00GHz 4 CPU
with 8GB of RAM.</p>
        <p>Under such conditions, we cannot compare systems on the basis of their speed.</p>
        <p>Reported figures are the average of 5 runs.</p>
        <p>Participation From the 21 systems participating to OAEI this year, 14 systems were
evaluated in this track. Several of these systems encountered problems: We
encountered problems with one very slow matcher (LogMapBio) that has been eliminated from
the pool of matchers. AML and ServOMBI had to be killed while they were unable to
match the second run of the energy data set. No timeout was explicitly set. We did not
investigate these problems.</p>
        <p>Compliance Table 4 synthesizes the results obtained by matchers.</p>
        <p>Globally results are far better on the biblio test than the energy one. This may be due
either to system overfit to biblio or to the energy dataset being erroneous. However, 5
systems obtained best overall F-measure on the energy data set (this is comparable to the
results obtained in 2014). It seems that run 1, 4 and 5 of energy generated ontologies
found erroneous by some parsers (the matchers did not return any results), but some
matchers where able to return relevant results. Curiously XMap did only work properly
on tests 2 and 3.</p>
        <p>Concerning F-measure results, all tested systems are above edna with LogMap-C
been lower (we excluded LogMapIM which is definitely dedicated to instance matching
only as well as JarvisOM and RSDLWD which outputed no useful results). Lily and
CroMatcher achieve impressive 90% and 88% F-measure. Not only these systems achieve
a high precision but a high recall of 83% as well. CroMatcher maintains its good results
on energy (while Lily cannot cope with the test), however LogMapLt obtain the best
F-measure (of 77%) on energy.</p>
        <p>Polarity We draw the triangle graphs for the biblio tests (Figure 1). It confirms that
systems are more precision-oriented than ever: no balaced system is visible in the middle
of the graph (only Mamba has a more balanced behavior).
This year, matcher performances have again reached their best level on biblio. However,
relaxation of constraints made many systems fail during the tests. Running on newly
generated tests has proved more difficult (but different systems fail on different tests).
Systems are still very oriented towards precision at the expense of recall.
4</p>
      </sec>
    </sec>
    <sec id="sec-4">
      <title>Anatomy</title>
      <p>The anatomy test case confronts matchers with a specific type of ontologies from the
biomedical domain. We focus on two fragments of biomedical ontologies which
describe the human anatomy6 and the anatomy of the mouse7. This data set has been used
since 2007 with some improvements over the years.
6 http://www.cancer.gov/cancertopics/cancerlibrary/</p>
      <p>terminologyresources/
7 http://www.informatics.jax.org/searches/AMA_form.shtml</p>
      <p>R=.9
R=.8 F=0.9 CroMatcher
We conducted experiments by executing each system in its standard setting and we
compare precision, recall, F-measure and recall+. The measure recall+ indicates the
amount of detected non-trivial correspondences. The matched entities in a non-trivial
correspondence do not have the same normalized label. The approach that generates
only trivial correspondences is depicted as baseline StringEquiv in the following section.</p>
      <p>We run the systems on a server with 3.46 GHz (6 cores) and 8GB RAM allocated
to each matching system. Further, we used the SEALS client to execute our evaluation.
However, we slightly changed the way precision and recall are computed, i.e., the results
generated by the SEALS client vary in some cases by 0.5% compared to the results
presented below. In particular, we removed trivial correspondences in the oboInOwl
namespace like:</p>
      <p>http://...oboInOwl#Synonym = http://...oboInOwl#Synonym
as well as correspondences expressing relations different from equivalence. Using the
Pellet reasoner we also checked whether the generated alignment is coherent, i.e., there
are no unsatisfiable concepts when the ontologies are merged with the alignment.
4.2</p>
      <sec id="sec-4-1">
        <title>Results</title>
        <p>In Table 5, we analyze all participating systems that could generate an alignment.
The listing comprises 15 entries. LogMap participated with different versions, namely
LogMap, LogMap-Bio, LogMap-C and a lightweight version LogMapLt that uses only
some core components. Similarly, DKP-AOM is also participating with two versions,
DKP-AOM and DKP-AOM-lite, DKP-AOM performs coherence analysis. There are
systems which participate in the anatomy track for the first time. These are COMMAND,
DKP-AOM, DKP-AOM-lite, GMap and JarvisOM. On the other hand, AML, LogMap (all
versions), RSDLWB and XMap participated in the anatomy track last year while Lily and
CroMatcher participated in 2011 and 2013 respectively. However, CroMatcher did not
produce an alignment within the given timeframe in 2013. For more details, we refer
the reader to the papers presenting the systems. Thus, this year we have 11 different
systems (not counting different versions) which generated an alignment.</p>
        <p>Three systems (COMMAND, GMap and Mamba) run out of memory and could
not finish execution with the allocated amount of memory. Therefore, they were run
on a different configuration with allocated 14 GB of RAM (Mamba additionally had
database connection problems). Therefore, the execution times for COMMAND and
GMap (marked with * and ** in the table) are not fully comparable to the other
systems. As last year, we have 6 systems which finished their execution in less than 100
seconds. The top systems in terms of runtimes are LogMap, RDSLWB and AML.
Depending on the specific version of the systems, they require between 20 and 40 seconds
to match the ontologies. The table shows that there is no correlation between quality
of the generated alignment in terms of precision and recall and required runtime. This
result has also been observed in previous OAEI campaigns.</p>
        <p>Table 5 also shows the results for precision, recall and F-measure. In terms of
Fmeasure, the top ranked systems are AML, XMap, LogMap-Bio and LogMap. The results
AML
XMap
LogMapBio
LogMap
GMap
CroMatcher
Lily
LogMapLt
LogMap-C
StringEquiv
DKP-AOM-lite
ServOMBI
RSDLWB
DKP-AOM
JarvisOM
COMMAND</p>
        <p>Runtime
of these four systems are at least as good as the results of the best systems in OAEI
2007-2010. AML, LogMap and LogMap-Bio produce very similar alignments compared
to the last years. For example, AML’s and LogMap’s alignment contained only one
correspondence less than the last year. Out of the systems which participated in the previous
years, only Lily showed improvement. Lily’s precision was improved from 0.81 to 0.87,
recall from 0.73 to 0.79 and the F-measure from 0.77 to 0.83. This is also the first time
that CroMatcher successfully produced an alignment given the set timeframe and its
result is 6th best with respect to the F-measure.</p>
        <p>This year we had 9 out of 15 systems which achieved an F-measure higher than the
baseline which is based on (normalized) string equivalence (StringEquiv in the table).
This is a slightly worse result (percentage-wise) than in the previous years when 7 out
of 10 (2014) and 13 out of 17 systems (2012) produced alignments with F-measure
higher than the baseline. The list of systems which achieved an F-measure lower than
the baseline is comprised mostly of newly competing systems. The only exception is
RSDLWB which competed last year when it also achieved a lower-than-baseline result.</p>
        <p>Moreover, nearly all systems find many non-trivial correspondences. Exceptions are
RSDLWB and DKP-AOM which generate only trivial correspondences.</p>
        <p>This year seven systems produced coherent alignments which is comparable to the
last year when 5 out of 10 systems achieved this.
This year we have again experienced an increase in the number of competing systems.
The list of competing systems is comprised of both systems which participated in the
previous years and new systems.</p>
        <p>The evaluation of the systems has shown that most of the systems which participated
in the previous years did not improve their results and in most cases they achieved
slightly worse results. The only exception is Lily which showed some improvement
compared to the previous time it competed. Out of the newly participating systems,
GMap displayed the best performance and achieved the 5th best result with respect to
the F-measure this year.
5</p>
      </sec>
    </sec>
    <sec id="sec-5">
      <title>Conference</title>
      <p>The conference test case requires matching several moderately expressive ontologies
from the conference organization domain.</p>
      <p>The data set consists of 16 ontologies in the domain of organizing conferences. These
ontologies have been developed within the OntoFarm project8.</p>
      <p>The main features of this test case are:
– Generally understandable domain. Most ontology engineers are familiar with
organizing conferences. Therefore, they can create their own ontologies as well as
evaluate the alignments among their concepts with enough erudition.
– Independence of ontologies. Ontologies were developed independently and based
on different resources, they thus capture the issues in organizing conferences from
different points of view and with different terminologies.
– Relative richness in axioms. Most ontologies were equipped with OWL DL axioms
of various kinds; this opens a way to use semantic matchers.</p>
      <p>Ontologies differ in their numbers of classes and properties, in expressivity, but also
in underlying resources.
5.2
We provide results in terms of F-measure, comparison with baseline matchers and
results from previous OAEI editions and precision/recall triangular graph based on sharp
reference alignment. This year we newly provide results based on the uncertain version
of reference alignment and on violations of consistency and conservativity principles.</p>
      <sec id="sec-5-1">
        <title>8 http://owl.vse.cz:8080/ontofarm/</title>
        <sec id="sec-5-1-1">
          <title>Evaluation based on sharp reference alignments We evaluated the results of partic</title>
          <p>ipants against blind reference alignments (labelled as rar2).9 This includes all pairwise
combinations between 7 different ontologies, i.e. 21 alignments.</p>
          <p>These reference alignments have been made in two steps. First, we have generated
them as a transitive closure computed on the original reference alignments. In order to
obtain a coherent result, conflicting correspondences, i.e., those causing
unsatisfiability, have been manually inspected and removed by evaluators. The resulting reference
alignments are labelled as ra2. Second, we detected violations of conservativity
using the approach from [34] and resolved them by an evaluator. The resulting reference
alignments are labelled as rar2. As a result, the degree of correctness and completeness
of the new reference alignment is probably slightly better than for the old one.
However, the differences are relatively limited. Whereas the new reference alignments are
not open, the old reference alignments (labeled as ra1 on the conference web page) are
available. These represent close approximations of the new ones.</p>
          <p>Prec. F0:5-m. F1-m. F2-m. Rec.</p>
          <p>Inc.Align. Conser.V. Consist.V.
9 More details about evaluation applying other sharp reference alignments are available at the
conference web page.
F0:5 weights precision higher than recall. The matchers shown in the table are ordered
according to their highest average F1-measure. We employed two baseline matchers.
edna (string edit distance matcher) is used within the benchmark test case and with
regard to performance it is very similar as the previously used baseline2 in the conference
track; StringEquiv is used within the anatomy test case. These baselines divide matchers
into three performance groups. Group 1 consists of matchers (AML, Mamba, LogMap-C,
LogMap, XMAP, GMap, DKP-AOM and LogMapLt) having better (or the same) results
than both baselines in terms of highest average F1-measure. Group 2 consists of
matchers (ServOMBI and COMMAND) performing better than baseline StringEquiv. Other
matchers (CroMatcher, Lily, JarvisOM and RSDLWB) performed slightly worse than
both baselines. The performance of all matchers regarding their precision, recall and
F1-measure is visualized in Figure 2. Matchers are represented as squares or triangles.
Baselines are represented as circles.</p>
          <p>Further, we evaluated performance of matchers separately on classes and properties.
We compared position of tools within overall performance groups and within only class
performance groups. We observed that on the one side ServOMBI and LogMapLt
improved their position in overall performance groups wrt. their position in only classes
performance groups due to their better property matching performance than baseline
edna. On the other side RSDLWB worsen its position in overall performance groups
wrt. its position in only classes performance groups due to its worse property matching
performance than baseline StringEquiv. DKP-AOM and Lily do not match properties at
all but they remained in their respective overall performance groups wrt. their positions
in only classes performance groups. More details about these evaluation modalities are
on the conference web page.</p>
          <p>Comparison with previous years wrt. ra2 Six matchers also participated in this test
case in OAEI 2014. The largest improvement was achieved by XMAP (recall from .44
to .51, while precision decreased from .82 to .81), and AML (precision from .80 to .81
and recall from .58 to .61). Since we applied rar2 reference alignment for the first time,
we used ra2, consistent but not conservativity violations free, reference alignment for
year-by-year comparison.</p>
          <p>Evaluation based on uncertain version of reference alignments The confidence
values of all correspondences in the sharp reference alignments for the conference track
are all 1.0. For the uncertain version of this track, the confidence value of a
correspondence has been set equal to the percentage of a group of people who agreed with
the correspondence in question (this uncertain version is based on reference alignment
labelled as ra1). One key thing to note is that the group was only asked to validate
correspondences that were already present in the existing reference alignments – so some
correspondences had their confidence value reduced from 1.0 to a number near 0, but
no new correspondence was added.</p>
          <p>There are two ways that we can evaluate matchers according to these “uncertain”
reference alignments, which we refer to as discrete and continuous. The discrete
evaluation considers any correspondence in the reference alignment with a confidence value
of 0.5 or greater to be fully correct and those with a confidence less than 0.5 to be fully</p>
          <p>F1-measure=0.7</p>
          <p>F1-measure=0.6
F1-measure=0.5
Mamba
LogMap-C
LogMap
XMAP
GMap
DKP-AOM
LogMapLt
ServOMBI
COMMAND
incorrect. Similarly, an matcher’s correspondence is considered a “yes” if the
confidence value is greater than or equal to the matcher’s threshold and a “no” otherwise.
In essence, this is the same as the “sharp” evaluation approach, except that some
correspondences have been removed because less than half of the crowdsourcing group
agreed with them. The continuous evaluation strategy penalizes an alignment system
more if it misses a correspondence on which most people agree than if it misses a more
controversial correspondence. For instance, if A B with a confidence of 0.85 in the
reference alignment and a matcher gives that correspondence a confidence of 0.40, then
that is counted as 0:85 0:40 = 0:34 of a true positive and 0:85{0:40 = 0:45 of a false
negative.</p>
          <p>
            The results from this year, see Table 7, follow the same general pattern as the
results from the 2013 systems discussed in [
            <xref ref-type="bibr" rid="ref5">5</xref>
            ]. Out of the 14 matchers, five (DKP-AOM,
JarvisOm, LogMapLt, Mamba, and RSDLWB) use 1.0 as the confidence values for all
correspondences they identify. Two (ServOMBI and XMap) of the remaining nine have
some variation in confidence values, though the majority are 1.0. The rest of the
matchers have a fairly wide variation of confidence values. Most of these are near the upper
end of the [
            <xref ref-type="bibr" rid="ref1">0,1</xref>
            ] range. The exception is Lily, which produces many correspondences
with confidence values around 0.5.
          </p>
          <p>Discussion In most cases, precision using the uncertain version of the reference
alignment is the same or less than in the sharp version, while recall is slightly greater with the
uncertain version. This is because no new correspondence was added to the reference
alignments, but controversial ones were removed.</p>
          <p>Regarding differences between the discrete and continuous evaluations using the
uncertain reference alignments, they are in general quite small for precision. This is
because of the fairly high confidence values assigned by the matchers. COMMAND’s
continuous precision is much lower because it assigns very low confidence values to
some correspondences in which the labels are equivalent strings, which many
crowdsourcers agreed with unless there was a compelling contextual reason not to. Applying
a low threshold value (0.53) for the matcher hides this issue in the discrete case, but the
continuous evaluation metrics do not use a threshold.</p>
          <p>Recall measures vary more widely between the discrete and continuous metrics. In
particular, matchers that set all confidence values to 1.0 see the biggest gains between
the discrete and continuous recall on the uncertain version of the reference alignment.
This is because in the discrete case incorrect correspondences produced by those
systems are counted as a whole false positive, whereas in the continuous version, they are
penalized a fraction of that if not many people agreed with the correspondence. While
this is interesting in itself, this is a one-time gain in improvement. Improvement on
this metric from year-to-year will only be possible if developers modify their systems
to produce meaningful confidence values. Another thing to note is the large drop in
Lily’s recall between the discrete and continuous approaches. This is because the
confidence values assigned by that alignment system are in a somewhat narrow range and
universally low, which apparently does not correspond well to human evaluation of the
correspondence quality.</p>
        </sec>
        <sec id="sec-5-1-2">
          <title>Evaluation based on violations of consistency and conservativity principles This</title>
          <p>year we performed evaluation based on detection of conservativity and consistency
violations [34]. The consistency principle states that correspondences should not lead to
unsatisfiable classes in the merged ontology; the conservativity principle states that
correspondences should not introduce new semantic relationships between concepts from
one of the input ontologies.</p>
          <p>Table 6 summarizes statistics per matcher. There are ontologies that have
unsatisfiable TBox after ontology merge (Uns.Ont.), total number of all conservativity principle
violations within all alignments (Conser.V.) and total number of all consistency
principle violations (Consist.V.).</p>
          <p>Five tools (AML, DKP-AOM, LogMap, LogMap-C and XMAP) do not violate
consistency. The lowest number of conservativity violations was achieved by LogMap-C
which has a repair technique for them. Four further tools have an average of
conservativity principle around 1 (DKP-AOM, JarvisOM, LogMap and AML).10 We should note
that these conservativity principle violations can be “false positives” since the
entailment in the aligned ontology can be correct although it was not derivable in the single
input ontologies.</p>
          <p>In conclusion, this year eight matchers (against five matchers last year for easier
reference alignment) performed better than both baselines on new, not only consistent
but also conservative, reference alignments. Next two matchers perform almost equally
well as the best baseline. Further, this year five matchers generate coherent alignments
(against four matchers last year). Based on uncertain reference alignments many more
10 All matchers but one delivered all 21 alignments. RSDLWB generated 18 alignments.
matchers provide alignments with a range of confidence values than in the past. This
evaluation modality will enable us to evaluate degree of convergence between this year’s
results and humans scores on the alignment task next years.
6</p>
        </sec>
      </sec>
    </sec>
    <sec id="sec-6">
      <title>Large biomedical ontologies (largebio)</title>
      <p>The largebio test case aims at finding alignments between the large and semantically
rich biomedical ontologies FMA, SNOMED-CT, and NCI, which contains 78,989,
306,591 and 66,724 classes, respectively.
The test case has been split into three matching problems: FMA-NCI, FMA-SNOMED
and SNOMED-NCI; and each matching problem in 2 tasks involving different
fragments of the input ontologies.</p>
      <p>
        The UMLS Metathesaurus [
        <xref ref-type="bibr" rid="ref3">3</xref>
        ] has been selected as the basis for reference
alignments. UMLS is currently the most comprehensive effort for integrating
independentlydeveloped medical thesauri and ontologies, including FMA, SNOMED-CT, and NCI.
Although the standard UMLS distribution does not directly provide alignments (in the
sense of [17]) between the integrated ontologies, it is relatively straightforward to
extract them from the information provided in the distribution files (see [21] for details).
      </p>
      <p>It has been noticed, however, that although the creation of UMLS alignments
combines expert assessment and auditing protocols they lead to a significant number of
logical inconsistencies when integrated with the corresponding source ontologies [21].</p>
      <p>Since alignment coherence is an aspect of ontology matching that we aim to
promote, in previous editions we provided coherent reference alignments by refining the
UMLS mappings using the Alcomo (alignment) debugging system [26], LogMap’s
(alignment) repair facility [20], or both [22].</p>
      <p>However, concerns were raised about the validity and fairness of applying
automated alignment repair techniques to make reference alignments coherent [30]. It is
clear that using the original (incoherent) UMLS alignments would be penalizing to
ontology matching systems that perform alignment repair. However, using automatically
repaired alignments would penalize systems that do not perform alignment repair and
also systems that employ a repair strategy that differs from that used on the reference
alignments [30].</p>
      <p>Thus, as in the 2014 edition, we arrived at a compromising solution that should be
fair to all ontology matching systems. Instead of repairing the reference alignments as
normal, by removing correspondences, we flagged the incoherence-causing
correspondences in the alignments by setting the relation to “?” (unknown). These “?”
correspondences will neither be considered as positive nor as negative when evaluating the
participating ontology matching systems, but will simply be ignored. This way, systems
that do not perform alignment repair are not penalized for finding correspondences that
(despite causing incoherences) may or may not be correct, and systems that do perform
alignment repair are not penalized for removing such correspondences.</p>
      <p>To ensure that this solution was as fair as possible to all alignment repair strategies,
we flagged as unknown all correspondences suppressed by any of Alcomo, LogMap or
AML [31], as well as all correspondences suppressed from the reference alignments of
last year’s edition (using Alcomo and LogMap combined). Note that, we have used the
(incomplete) repair modules of the above mentioned systems.</p>
      <p>The flagged UMLS-based reference alignment for the OAEI 2015 campaign is
summarized in Table 8.</p>
      <sec id="sec-6-1">
        <title>Evaluation setting, participation and success</title>
        <p>We have run the evaluation in a Ubuntu Laptop with an Intel Core i7-4600U CPU @
2.10GHz x 4 and allocating 15Gb of RAM. Precision, Recall and F-measure have been
computed with respect to the UMLS-based reference alignment. Systems have been
ordered in terms of F-measure.</p>
        <p>In the OAEI 2015 largebio track, 13 out of 22 participating OAEI 2015 systems have
been able to cope with at least one of the tasks of the largebio track. Note that
RiMOMIM, InsMT+, STRIM, EXONA, CLONA and LYAM++ are systems focusing on either the
instance matching track or the multifarm track, and they did not produce any alignment
for the largebio track. COMMAND and Mamba did not finish the smallest largebio task
within the given 12 hours timeout, while GMap and JarvisOM gave an “error exception”
when dealing with the smallest largebio task.
6.3</p>
      </sec>
      <sec id="sec-6-2">
        <title>Background knowledge</title>
        <p>Regarding the use of background knowledge, LogMap-Bio uses BioPortal as mediating
ontology provider, that is, it retrieves from BioPortal the most suitable top-10 ontologies
for the matching task.</p>
        <p>LogMap uses normalisations and spelling variants from the general (biomedical)
purpose UMLS Lexicon.</p>
        <p>AML has three sources of background knowledge which can be used as mediators
between the input ontologies: the Uber Anatomy Ontology (Uberon), the Human
Disease Ontology (DOID) and the Medical Subject Headings (MeSH).</p>
        <p>XMAP has been evaluated with two variants: XMAP-BK and XMAP. XMAP-BK uses
synonyms provided by the UMLS Metathesaurus, while XMAP has this feature
deactivated. Note that matching systems using UMLS-Metathesaurus as background
Fig. 3. The Y axis depicts the time intervals between the requests to the user/oracle (whiskers:
Q1-1,5IQR, Q3+1,5IQR, IQR=Q3-Q1). The labels under the system names show the average
number of requests and the mean time between the requests for the three runs.
milliseconds respectively. On the other hand, while the requests periods for ServOMBI
are under 10 ms in most of the cases we see that there are some outliers requiring more
than a second. Furthermore a manual inspection of the intervals showed that in several
cases it takes more than 10 seconds between the questions to the user and in one
extreme case—250 seconds. It can also be seen that the requests intervals for this system
increase at the last 50–100 questions. JarvisOM displays a delay in its requests in
comparison to the other systems. The average interval at which a question is presented to
the user is 1 second with about half of the requests to the user taking more than 1,5
seconds. However it issues the quetions during the alignemnt process and not as a post
processing step.</p>
        <p>The take away of this analyses is the large improvement for JarvisOM in all
measures and error rates with respect to its non-interactive results. The growth of the error
rate impacts different measures in the different systems. The effect of introducing
interactions with the oracle/user is mostly pronounced for the precision measure - the
precision for all systems (except AML) in the different error rates is higher than their
precision in the evaluation of the non-interactive Anatomy track.</p>
      </sec>
      <sec id="sec-6-3">
        <title>Results for the conference dataset</title>
        <p>Tables 20, 21, 22 and 23 below present the results for the Conference dataset with four
different error rates. The ”Precision Oracle”, ”Recall Oracle” and ”F-measure Oracle”
columns contain the evaluation results ”according to the oracle”, meaning against the
oracle’s alignment (i.e., the reference alignment as modified by the randomly introduced
errors). Figure 4 shows the average requests intervals per task (21 tasks in total per run)
between the questions to the user/oracle for the different systems and error rates for all
tasks and the three runs (the runs are depicted with different colors). The first number
under the system names is the average number of requests and the second number is the
average period of the average requests intervals for all tasks and runs.</p>
        <p>We first focus on the performance of the systems with an all-knowing oracle
(Table 20). In this case, all systems improve their results compared to the non-interactive
version of the Conference track. The biggest improvement in F-measure is achieved
by ServOMBI with 23 percentage points. Other systems also show substantial
improvements, AML improves the F-measure by 8, JarvisOM by 13 and LogMap by around 4
percentage points. Closer inspection shows that for different systems the improvement of
F-measure can be attributed to different factors. For example, in the case of ServOMBI
and LogMap interaction with the user improved precision while recall experienced only
slight improvement. On the other hand, JarvisOM improved recall substantially while
keeping similar level of precision. Finally, AML improved precision by 10 and recall by
6 percentage points which contributed to a higher F-measure.</p>
        <p>As expected, the results start deteriorating when introducing the error in the oracle’s
answers. Interestingly, even with the error rate of 0.3 (Table 23) most systems perform
similar (with respect to the F-measure) to their non-interactive version. For example,
AML’s F-measure in the case with 0.3 error rate is only 1 percentage point worse than
the non-interactive one. The most substantial difference is in the case of ServOMBI
with an oracle with the error rate of 0.3 where the system achieves around 5 percentage
points worse result w.r.t. F-measure than in the non-interactive version. Again closer
inspection shows that different systems are affected in different ways when errors are
introduced. For example, if we compare the 0.0 and 0.3 case, we can see that for AML,
precision is affected by 11 and recall by 6 percentage points. In the case of JarvisOM,
precision drops by 19 while recall drops by only 4 percentage points. LogMap is
affected in a similar manner and its precision drops by 9 while the recall drops by only
3 percentage points. Finally, the most substantial change is in the case of ServOMBI
where the precision drops from 100% to 66% and the recall shows a drop of 22
percentage points. Like in the Anatomy dataset, LogMap and ServOMBI also show a drop
in performance in relation to the oracle’s reference with the increase of the error rate,
which indicates a supralinear impact of the errors. AML again shows a constant
performance that reflects a linear impact of the errors. Surprisingly, JarvisOM also shows a
constant performance, which is a different behavior than in the anatomy case.</p>
        <p>When it comes to the number of request to the oracle, 3 out of 4 systems do around
150 requests while ServOMBI does most requests, namely 550. AML, JarvisOM and
LogMap do not repeat their requests while around 40% of requests done by ServOMBI
are repeated requests. Across the three runs and different error rates the AML and
LogMap mean times between requests for all tasks are less than 3 ms. On the other
.toT seq</p>
        <p>R
e a
R r</p>
        <p>O
TN .940 .1160 .100 .130</p>
        <p>5 9
P .0 .0 .0 .0
.0 .0 .
l L O a M
o s
o M i
T A rv</p>
        <p>M O
g v
o r
Ja L e</p>
        <p>S
e
l
c
a
r
o
t
c
e
f
r
e
p
–
t
e
s
a
t
a
d
e
c
n
e
r
e
f
n
o
C
.
0
2
e
l
PF .78 .013 .123 .167
.3 .0 .7 .3
7 7 9</p>
        <p>4
4 5 5 9
1 1 1 2
.3 .0 .7 .3
7 4 7 5
4 5 5 5
1 1 1 5
e o
Rn
.
c 1 3 0 7
e .7 .5 .6 .5
R 0 0 0 0
M p B</p>
        <p>M p B</p>
        <p>M p B
l L O a M
o s
o M i
T A rv
e
c
n
e
r
e
f
n
o
C
.
3
2
e
l
b
a
T
b .</p>
        <p>T -mon
a</p>
        <p>Fn
.</p>
        <p>b .
a
T -mon</p>
        <p>Fn
FN .103 .63 .20 .40</p>
        <p>1 3
P .3 .7 .0 .0
TN .763 .973 .847 .1107
P .0 .7 .0 .0
.0 .0 .7 .7
9 5 8 5
4 5 5 9
1 1 1 2
.0 .0 .7 .7
9 5 8 4
4 5 5 5
1 1 1 5
c 9 2 9 0
e .6 .5 .5 .5
R 0 0 0 0</p>
        <p>7 8 9 1
-m .7 .5 .6 .6</p>
        <p>F 0 0 0 0
c n
rePno
FN .207 .100 .143 .351
P .7 .0 .3 .0
Fig. 4. The Y axis depicts the average time between the requests per task in the Conference dataset
(whiskers: Q1-1,5IQR, Q3+1,5IQR, IQR=Q3-Q1). The labels under the system names show the
average number of requests and the mean time between the requests (calculated by taking the
average of the average request intervals per task) for the three runs and all tasks.
hand, mean time between requests for ServOMBI and JarvisOM are around 30 and 10
ms respectively. While in most cases there is little to no delay between requests, there
are some outliers. These are most prominent for ServOMBI where some requests were
delayed for around 2 seconds which is substaintally longer than the mean.</p>
        <p>This year we have two systems, AML and LogMap, which competed in the last year’s
campaign. When comparing to the results of last year (perfect oracle), AML improved its
F-measure by around 2 percentage points. This increase can be accounted to increased
precision (increase of around 3 percentage points). On the other hand, LogMap shows a
slight decrease in recall and precision, and hence, in F-measure.
8.6</p>
      </sec>
      <sec id="sec-6-4">
        <title>Results for the largebio dataset</title>
        <p>Tables 24, 25, 26 and 27 below present the results for the largebio dataset with four
different error rates. The “precision oracle”, “recall oracle” and “F-measure oracle”
columns contain the evaluation results “according to the oracle”, meaning against the
oracle’s alignment, i.e., the reference alignment as modified by the randomly introduced
errors. Figure 5 shows the average requests intervals per task (6 tasks in total) between
the questions to the user/oracle for the different systems and error rates for all tasks
and a single runs. The first number under the system names is the average number of
requests and the second number is the average period of the average requests intervals
for all tasks in the run.</p>
        <p>Of the four systems participating in this track this year, only AML and LogMap
were able to complete the full largebio dataset. ServOMBI was only able to match the
FMA-NCI small fragments and FMA-SNOMED small fragments, whereas JarvisOM
was unable to complete any of the tasks. Therefore, ServOMBI’s results are partial, and
not directly comparable with those of the other systems (marked with * in the results
table and Figure 5).</p>
        <p>With an all-knowing oracle (Table 24), AML, LogMap and ServOMBI all improved
their performance in comparison with the non-interactive version of the largebio track.
The biggest improvement in F-measure was achieved by LogMap with 4, followed by
AML with 3, then ServOMBI with 2 percentage points. AML showed the greatest
improvement in terms of recall, but also increased its precision substantially; LogMap had
the greatest improvement in terms of precision, but also showed a significant increase
in recall; and ServOMBI improved essentially only with regard to precision, obtaining
100% as in the other datasets.</p>
        <p>The introduction of (simulated) user errors had a very different effect on the three
systems: AML shows a slight drop in performance of 3 percentage points in F-measure
between 0 and 0.3 error rate (Table 27), and is only slightly worse than its
noninteractive version at 0.3 error rate; LogMap shows a more pronounced drop of 6
percentage points in F-measure; and ServOMBI shows a substantial drop of 17 percentage
points in F-measure. Unlike in the other datasets, all systems are affected significantly
by the error with regard to both precision and recall. Like in the other datasets, AML
shows a constant performance in relation to the oracle’s reference, indicating a linear
impact of the errors, whereas the other two systems decrease in performance as the error
increases, indicating a supralinear impact of the errors.</p>
        <p>Regarding the number of request to the oracle, AML was the more sparing system,
with only 10,217, whereas LogMap made almost three times as many requests (27,436).
ServOMBI was again the more inquisitive system, with 21,416 requests on only the two
smallest tasks in the dataset (for comparison, AML made only 1,823 requests on these
two tasks and LogMap made 6,602). As in the other datasets, ServOMBI was the only
system to make redundant requests to the oracle. Interestingly, both LogMap and
ServOMBI increased the number of requests with the error, whereas AML had a constant
number of requests. Figure 5 presents a comparison between the systems regarding the
average time periods for all tasks at which the system presents a question to the user.
Across the different error rates the average requests intervals for all tasks for AML and
LogMap are around 0 millisecond. For ServOMBI they are slightly higher (25
milliseconds on average) but a manual inspection of the results shows some intervals larger than
1 second (often those are between some of the last requests the system performs).
8.7</p>
      </sec>
      <sec id="sec-6-5">
        <title>Discussion</title>
        <p>This year is the first time we have considered a non-perfect domain expert, i.e., a
domain expert which can provide wrong answers. As expected, the performance of the</p>
        <p>F
rac -m</p>
        <p>T
a
b
l
e
2
6
.</p>
        <p>L
a
r
g
e
b
i
o
d
a
t
a
s
e
t
–
e
r
r
o
r
r
a
t
e
.
2
e L
rvO ogM AM oT
M o</p>
        <p>a L l
B p
I
*
0 0 0 P
.9 .9 .9 re
9 2 2 c</p>
        <p>F
rac -m</p>
        <p>O
r R
a e
c c
le .</p>
        <p>R
seq .toT
.</p>
        <p>R D
eq is
.s .t
e L
rvO ogM AM oT
M o</p>
        <p>a L l
B p
I
*
1 0 0 P
.0 .9 .9 re
0 4 3 c
B p
I
*
0 0 0 P
.9 .9 .9 re
8 0 1
1 4 1
8 0 3 F
7 1 0 N
4 5 6
.
3
0 4 7
0 0 7
0
0</p>
        <p>0
O</p>
        <p>O
3471 28416 9461 PT
192 6269 1049 FP
1257 2764 891 FN
e L
rvO ogM AM oT
M</p>
        <p>o
a L l
B p
I
*
1 0 0 P
.0 .9 .9 re
0 7 4 c
0 0 0
T .8 .7 .8
a 3 9 2
b
l
e
2
4
.</p>
        <p>L
a
r
g
e
b
i
o
d
a
t
a
s
e
t
–
p
e
r
f
e
c
t
o
r
a
c
l
e
systems deteriorated with the increase of the error rate. However, an interesting
observation is that the errors had different impact on different systems reflecting the different
interactive strategies employed by the systems. In some cases, erroneous answers from
the oracle had the highest impact on the recall, in other cases on the precision, and in
others still both measures were significantly affected. Also interesting is the fact that the
impact of the errors was linear in some systems and supralinear in others, as reflected
by their performance in relation to the oracle’s alignment. A supralinear impact of the
errors indicates that the system is making inferences from the user and thus deciding on
the classification of multiple correspondence candidates based on user feedback about
only one correspondence. This is an effective strategy for reducing the burden on the
user, but alas leaves the matching system more susceptible to user errors. An extreme
example of this is JarvisOM on the Anatomy dataset, as it uses an active-learning
approach based on solely 7 user requests, and consequently is profoundly affected when
faced with user errors given the size of the Anatomy dataset alignment. Curiously, this
system behaves very differently in the Conference dataset, showing a linear impact of
the errors, as in this case 7 requests (which is the average number it makes per task)
represent a much more substantial portion of the Conference alignments ( 50%) and
thus leads to less inferences and consequently less impact of errors.</p>
        <p>Apart from JarvisOM, all the systems make use of user interactions exclusively in
post-matching steps to filter their candidate correspondences. LogMap and AML both
request feedback on only selected correspondence candidates (based on their similarity
patterns or their involvement in unsatisfiabilities). By contrast, ServOMBI employs the
user to validate all its correspondence candidates (after two distinct matching stages),
which corresponds to user validation rather than interactive matching. Consequently, it
makes a much greater number of user requests than the other systems, and in being the
system most dependent on the user, is also the one most affected by user errors.</p>
        <p>With regard still to the number of user requests, it is interesting to note that both
ServOMBI and LogMap generally increased the number of requests with the error, whereas
AML and JarvisOM kept their number approximately constant. The increase is
natural, as user errors can lead to more complex decision trees when interaction is used in
filtering steps and inferences are drawn from the user feedback (such as during
alignment repair) which leads to an increased number of subsequent requests. JarvisOM is
not affected by this because it uses interaction during matching and makes a fixed 7-8
requests per matching task, whereas AML prevents it by employing a maximum query
limit and stringent stopping criteria.</p>
        <p>
          Two models for system response times are frequently used in the literature [
          <xref ref-type="bibr" rid="ref7">7</xref>
          ]:
Shneiderman and Seow take different approaches to categorize the response times.
Shneiderman takes task-centered view and sort out the response times in four categories
according to task complexity: typing, mouse movement (50-150 ms), simple frequent
tasks (1 s), common tasks (2-4 s) and complex tasks (8-12 s). He suggests that the user
is more tolerable to delays with the growing complexity of the task at hand.
Unfortunately no clear definition is given for how to define the task complexity. The Seow’s
model looks at the problem from a user-centered perspective by considering the user
expectations towards the execution of a task: instantaneous (100-200 ms), immediate
(0.5-1 s), continuous (2-5 s), captive (7-10 s); Ontology matching is a cognitively
demanding task and can fall into the third or forth categories in both models. In this regard
the response times (request intervals as we call them above) observed with the Anatomy
dataset (with the exception of several measurements for ServOMBI) fall into the
tolerable and acceptable response times in both models. The same applies for the average
requests intervals for the 6 tasks in the largebio dataset. The average request intervals
for the Conference dataset are lower (with the exception of ServOMBI) than those
discussed for the Anatomy dataset. It could be the case however that the user could not
take advantage of very low response times because the task complexity may result in
higher user response time (analogically it measures the time the user needs to respond
to the system after the system is ready).
9
        </p>
      </sec>
    </sec>
    <sec id="sec-7">
      <title>Ontology Alignment For Query Answering (OA4QA)</title>
      <p>Ontology matching systems rely on lexical and structural heuristics and the integration
of the input ontologies and the alignments may lead to many undesired logical
consequences. In [21], three principles were proposed to minimize the number of potentially
unintended consequences, namely: (i) consistency principle, the alignment should not
lead to unsatisfiable classes in the integrated ontology; (ii) locality principle, the
correspondences should link entities that have similar neighborhoods; (iii) conservativity
principle, the alignments should not introduce alterations in the classification of the
input ontologies. The occurrence of these violations is frequent, even in the reference
alignments sets of the Ontology Alignment Evaluation Initiative (OAEI) [35, 36].</p>
      <p>Violations to these principles may hinder the usefulness of ontology matching. The
practical effect of these violations, however, is clearly evident when ontology
alignments are involved in complex tasks such as query answering [26]. The traditional
tracks of OAEI evaluate ontology matching systems w.r.t. scalability, multi-lingual
support, instance matching, reuse of background knowledge, etc. Systems’ effectiveness is,
however, only assessed by means of classical information retrieval metrics, i.e.,
precision, recall and F-measure, w.r.t. a manually-curated reference alignment, provided by
the organizers. The OA4QA track [37], introduced in 2015, evaluates these same
metrics, with respect to the ability of the generated alignments to enable the answer of a set
of queries in an ontology-based data access (OBDA) scenario, where several ontologies
exist. Our target scenario is an OBDA scenario where one ontology provides the
vocabulary to formulate the queries (QF-Ontology) and the second is linked to the data and
it is not visible to the users (DB-Ontology). Such OBDA scenario is presented in
realworld use cases, e.g., the Optique project15 [19, 24, 35]. The integration via ontology
alignment is required since only the vocabulary of the DB-Ontology is connected to the
data. OA4QA will also be key for investigating the effects of logical violations
affecting the computed alignments, and evaluating the effectiveness of the repair strategies
employed by the matchers.
The set of ontologies coincides with that of the conference track (x5), in order to
facilitate the understanding of the queries and query results. The dataset is however extended
with synthetic ABoxes, extracted from the DBLP dataset.16</p>
      <p>Given a query q expressed using the vocabulary of ontology O1, another
ontology O2 enriched with synthetic data is chosen. Finally, the query is executed over the
aligned ontology O1 [ M [ O2, where M is an alignment between O1 and O2. Here
O1 plays the role of QF-Ontology, while O2 that of DB-Ontology.
The considered evaluation engine is an extension of the OWL 2 reasoner HermiT, known
as OWL-BGP17 [25]. OWL-BGP is able to process SPARQL queries in the
SPARQLOWL fragment, under the OWL 2 Direct Semantics entailment regime [25]. The queries
employed in the OA4QA track are standard conjunctive queries, that are fully supported
by the more expressive SPARQL-OWL fragment. SPARQL-OWL, for instance, also
15 http://www.optique-project.eu/
16 http://dblp.uni-trier.de/xml/
17 https://code.google.com/p/owl-bgp/
support queries where variables occur within complex class expressions or bind to class
or property names.</p>
      <sec id="sec-7-1">
        <title>Evaluation metrics and gold standard</title>
        <p>The evaluation metrics used for the OA4QA track are the classic information retrieval
ones, i.e., precision, recall and F-measure, but on the result set of the query evaluation.
In order to compute the gold standard for query results, the publicly available reference
alignments ra1 has been manually revised. The aforementioned metrics are then
evaluated, for each alignment computed by the different matching tools, against the ra1, and
manually repaired version of ra1 from conservativity and consistency violations, called
rar1 (not to be confused with ra2 alignment of the conference track).</p>
        <p>Three categories of queries are considered in OA4QA: (i) basic queries: instance
retrieval queries for a single class or queries involving at most one trivial correspondence
(that is, correspondences between entities with (quasi-)identical names), (ii) queries
involving (consistency or conservativity) violations, (iii) advanced queries involving
nontrivial correspondences.</p>
        <p>For unsatisfiable ontologies, we tried to apply an additional repair step, that
consisted in the removal of all the individuals of incoherent classes. In some cases, this
allowed to answer the query, and depending on the classes involved in the query itself,
sometimes it did not interfere in the query answering process.</p>
      </sec>
      <sec id="sec-7-2">
        <title>9.4 Impact of the mappings in the query results</title>
        <p>The impact of unsatisfiable ontologies, related to the consistency principle, is
immediate. The conservativity principle, compared to the consistency principle, received less
attention in literature, and its effects in a query answering process is probably less
known. For instance, consider the aligned ontology OU computed using confof and
ekaw as input ontologies (Oconfof and Oekaw, respectively), and the ra1 reference
alignment between them. OU entails ekaw:Student v ekaw:Conf P articipant,
while Oekaw does not, and therefore this represents a conservativity principle
violation [35]. Clearly, the result set for the query q(x) ekaw:Conf P articipant(x)
will erroneously contain any student not actually participating at the conference. The
explanation for this entailment in OU is given below, where Axioms 1 and 3 are
correspondences from the reference alignment.</p>
        <p>conf of :Scholar</p>
        <p>ekaw:Student
conf of :Scholar v conf of :P articipant
conf of :P articipant
ekaw:Conf P articipant
(1)
(2)
(3)
In what follows, we provide possible (minimal) alignment repairs for the
aforementioned violation:
– the weakening of Axiom 1 into conf of :Scholar w ekaw:Student,
– the weakening of Axiom 3 into conf of :P articipant w ekaw:Conf P articipant.</p>
        <p>Repair strategies could disregard weakening in favor of complete correspondence
removal, in this case the removal of either Axiom 1, or Axiom 3 could be possible
repairs. Finally, for strategies including the input ontologies as a possible repair target,
the removal of Axiom 2 can be proposed as a legal solution to the problem.
Table 28 shows the average precision, recall and f-measure results for the whole set of
queries. Matchers are evaluated on 18 queries in total, for which the sum of expected
answers is 1724. Some queries have only 1 answer while other have as many as 196.
AML, DKPAOM, LogMap, LogMap-C and XMap were the only matchers whose
alignments allowed to answer all the queries of the evaluation.</p>
        <p>AML was the best performing tool for what concerns averaged precision (same value
as XMAP), recall (same value as LogMap) and F-measure, closely followed by LogMap,
LogMap-C and XMap.</p>
        <p>Considering Table 28, the difference in results between the publicly available
reference alignment of conference track (ra1) and its repaired version (rar1, not to be
confused with ra2 of the conference track) was not significant. The F-measure ranking
between the two reference alignments is almost totally preserved, the only notable
variation concerns Lily, which is ranked 11th w.r.t. ra1, and 9th w.r.t. rar1 (improving its
results w.r.t. GMap and LogMapLt).</p>
        <p>If we compare Table 28 (the results of the present track) and Table 6, page 14 (w.r.t.
the results of conference track) we can see that 3 out of4 matchers in the top-4
ranking are shared, even if the ordering is different. Considering rar1 alignment, the gap
between the best performing matchers and the others is highlighted, and it also allows
to differentiate more among the least performing matchers, and seems therefore more
suitable as a reference alignment in the context of the OA4QA track evaluation.</p>
        <p>Comparing Table 28 to Table 6 for what concerns the logical violations of the
different matchers participating at the conference track, it seems that a negative correlation
between the ability of answering queries and the average degree of incoherence of the
matchers exists. For instance, taking into account the different positions in the ranking
of LogMapLt (the version of LogMap not equipped with logical repair facilities), we can
see that it is penalized more in our test case than in the traditional conference track, due
to its target scenario. ServOMBI, instead, even if presenting many violations and even
if most of its alignment is suffering from incoherences, is in general able to answer
enough of the test queries (6 out of 18).</p>
        <p>LogMapC, to the best of our knowledge the only ontology matching systems fully
addressing conservativity principle violations, did not outperform LogMap, because
some correspondences removed by its extended repair capabilities prevented to answer
one of the queries (the result set was empty as an effect of correspondence removal).
Alignment repair does not only affect precision and recall while comparing the
computed alignment w.r.t. a reference alignment, but it can enable or prevent the capability
of an alignment to be used in a query answering scenario. As experimented in the
evaluation, the conservativity violations repair technique of LogMapC on one hand improved
its performances on some queries w.r.t. LogMap matcher, but in one cases it actually
prevented to answer a query due to a missing correspondence. This conflicting effect
in the process of query answering imposes a deeper reflection on the role of ontology
alignment debugging strategies, depending on the target scenario, similarly to what
already discussed in [30] for incoherence alignment debugging.</p>
        <p>The results we presented depend on the considered set of queries. What clearly
emerges is that the role of logical violations is playing a major role in our evaluation,
and a possible bias due to the set of chosen queries can be mitigated by an extended set
of queries and synthetic data. We hope that this will be useful in the further exploration
of the findings of this first edition of the OA4QA track.</p>
        <p>As a final remark, we would like to clarify that the entailment of new knowledge,
obtained using the alignments, is not always negative, and conservativity principle
violations can be false positives. Another extension to the current set of queries would
target such false positives, with the aim of penalizing the indiscriminate repairs in
presence of conservativity principle violations.
10</p>
      </sec>
    </sec>
    <sec id="sec-8">
      <title>Instance matching</title>
      <p>The instance matching track aims at evaluating the performance of matching tools
identify relations between pairs of items/instances found in Aboxes. The track is
organized in five independent tasks, namely author disambiguation (author-dis task), author
recognition (author-rec task), value semantics (val-sem task), value structure (val-struct
task), and value structure semantics (val-struct-sem task).</p>
      <p>Each task is articulated in two tests, namely sandbox and mainbox, with different
scales, i.e., number of instances to match:
– Sandbox (small scale) is an open test, meaning that the set of expected mappings,
i.e., reference alignment, is given in advance to the participants.
– Mainbox (medium scale) is a blind test, meaning that the reference alignment is not
given in advance to the participants.</p>
      <p>Each test contains two datasets called source and target and the goal is to discover
the matching pairs, i.e., mappings, among the instances in the source dataset and the
instances in the target dataset.</p>
      <p>For the sake of clarity, we split the presentation of task results in two different
sections as follows.</p>
      <sec id="sec-8-1">
        <title>Results for author disambiguation (author-dis) and author recognition (author-rec) tasks</title>
        <p>The goal of author-dis and author-rec tasks is to discover links between pairs of OWL
instances referring to the same person, i.e., author, based on their publications. In both
tasks, expected mappings are 1:1 (one person of the source dataset corresponds to
exactly one person of the target dataset and vice versa).</p>
        <p>About the author-dis task, in both source and target datasets, authors and
publications are described as instances of the classes http://islab.di.unimi.it/
imoaei2015#Person and http://islab.di.unimi.it/imoaei2015#
Publication, respectively. Publications are associated with the
corresponding person instance through the property http://islab.di.unimi.it/
imoaei2015#author_of. Author and publication information are differently
described in the two datasets. For example, only the first letter of author names and the
initial part of publication titles are shown in the target dataset while the full strings
are provided in the source datasets. The matching challenge regards the capability to
resolve such a kind of ambiguities on author and publication descriptions.</p>
        <p>About the author-rec task, author and publication descriptions in the source dataset
are analogous to those in the author-dis task. As a difference, in the target dataset, each
author/person is only associated with a publication titled “Publication report”
containing aggregated information, such as number of publications, h-index, years of activity,
and number of citations. The matching challenge regards the capability to link a person
in the source dataset with the person in the target dataset containing the corresponding
publication report.</p>
        <p>Participants to author-dis and author-rec tasks are EXONA, InsMT+, Lily, LogMap,
and RiMOM. Results are shown in Table 29 and 30, respectively.</p>
        <p>For each tool, we provide the number of mapping expected in the ground truth, the
number of mapping actually retrieved by the tool, and tool performances in terms of
precision, recall, and F-measure.</p>
        <p>On the author-dis task, we note that good results in terms of precision and recall
are provided by all the participating tools. As a general remark, precision values are
slightly better than recall values. This behavior highlights the consolidated maturity of</p>
        <p>EXONA
InsMT+
Lily
LogMap
RiMOM
854
854
854
854
854</p>
        <p>Exp. mappings Retr. mappings Prec. F-m. Rec.</p>
        <p>EXONA
InsMT+
Lily
LogMap
RiMOM
854
854
854
854
854
instance matching tools when the alignment goal is to handle syntax modifications in
instance descriptions. On the author-rec task, the differences in tool performances are
more marked. In particular, we note that Lily, LogMap, and RiMOM have better results
than EXONA and InsMT+. Probably, this is due to the fact that the capability to align the
summary publication report to the appropriate author requires reasoning functionalities
that are available to only a subset of the participating tools. The distinction between
sandbox and mainbox tests puts in evidence that the capability to handle large-scale
datasets is complicated for most of the participating tools. We note that LogMap and
RiMOM are the best performing tools on the mainbox tests, but very-long execution
times usually characterize participants in the execution of large-scale tests. We argue
that this is a forthcoming challenging issue in the field of instance matching, on which
further experimentations and tests need to focus in the future competitions.</p>
      </sec>
      <sec id="sec-8-2">
        <title>Results for value semantics (val-sem), value structure (val-struct), and value structure semantics (val-struct-sem) tasks</title>
        <p>The val-sem, val-struct, and val-struct-sem tasks are three evaluation tasks of instance
matching tools where the goal is to determine when two OWL instances describe the
same real world object. The datasets have been produced by altering a set of source
data and generated by SPIMBENCH [32] with the aim to generate descriptions of the
same entity where value-based, structure-based and semantics-aware transformations
are employed in order to create the target data. The value-based transformations
consider mainly typographical errors and different data formats, the structure-based
transformations consider transformations applied on the structure of object and datatype
properties and the semantics-aware transformations are transformations at the instance
level considering the schema. The latter are used to examine if the matching systems
take into account RDFS and OWL constructs in order to discover correspondences
between instances that can be found only by considering schema information.</p>
        <p>We stress that an instance in the source dataset can have none or one matching
counterpart in the target dataset. A dataset is composed of a Tbox and a corresponding
Abox. Source and target datasets share almost the same Tbox (with some difference in
the properties’ level, due to the structure-based transformations). Ontology is described
through 22 classes, 31 datatype properties, and 85 object properties. From those
properties, there is 1 an inverse functional property and 2 are functional properties. The
sandbox scale is 10K instances while the mainbox scale is 100K instances.</p>
        <p>We asked the participants to match the Creative Works instances (NewsItem,
BlogPost and Programme) in the source dataset against the instances of the corresponding
class in the target dataset. We expected to receive a set of links denoting the pairs of
matching instances that they found to refer to the same entity. The datasets of the
valsem task have been produced by altering a set of source data through value-based and
semantics-aware transformations, while val-struct through value-based and
structurebased transformations and val-struct-sem task through value-based, structure-based and
semantics-aware.</p>
        <p>The participants to these tasks are LogMap and STRIM. For evaluation, we built a
ground truth containing the set of expected links where an instance i1 in the source
dataset is associated with an instance in the target dataset that has been generated as an
altered description of i1.</p>
        <p>The way that the transformations were done, was to apply value-based,
structurebased and semantics-aware transformations, on different triples pertaining to one class
instance. For example, regarding the val-struct task, for an instance u1, we performed
a value-based transformation on its triple (u1, p1, o1) where p1 is a data type property
and a structure-based transformation on its triple (u1, p2, o2).</p>
        <p>The evaluation has been performed by calculating precision, recall, and F-measure
and results are provided in Tables 31, 32, 33.</p>
        <p>The main comment is that the quality of the results for both LogMap and STRIM is
very high as we created the tasks val-sem, val-struct, and val-struct-sem in order to be
the easiest ones. LogMap and STRIM have consistent behavior for the sandbox and the
mainbox tasks, a fact that shows that both systems can handle different sizes of data
without reducing their performance.</p>
        <p>LogMap’s performance drops for tasks that consider structure-based
transformations (val-struct and val-struct-sem). Also, it produces links that are quite often correct
(resulting in a good precision) but fails in capturing a large number of the expected
links (resulting in a lower recall). STRIM’s performance drops for tasks that consider
semantics-aware transformations (val-sem and val-struct-sem) as expected. The
probability of capturing a correct link is high, but the probability of a retrieved link to be
correct is lower, resulting in a high recall but not equally high precision.</p>
        <p>Exp. mappings Retr. mappings Prec. F-m. Rec.</p>
        <p>STRIM
LogMap
9649
9649</p>
        <p>Exp. mappings Retr. mappings Prec. F-m. Rec.</p>
        <p>STRIM
LogMap
10601
10601</p>
        <p>Sandbox task
10657
8779</p>
        <p>Exp. mappings Retr. mappings Prec. F-m. Rec.</p>
        <p>STRIM
LogMap
9790
9790</p>
        <p>Sandbox task
10639
7779</p>
      </sec>
    </sec>
    <sec id="sec-9">
      <title>Lesson learned and suggestions</title>
      <p>Here are lessons learned from running OAEI 2015:
A) This year indicated again that requiring participants to implement a minimal
interface was not a strong obstacle to participation with some exceptions. Moreover,
the community seems to get used to the SEALS infrastructure introduced for OAEI
2011.</p>
      <p>B) It would be useful to tighten the rules for evaluation so that the we can again write
that “All tests have been run entirely from the SEALS platform with the strict same
protocol” and we do not end up with one evaluation setting tailored for each system.
This does not mean that we should come back to the exact setting of two years ago,
but that evaluators and tool developers should decide for one setting and stick to it
(i.e. avoid system variants participating only in a concrete track).</p>
      <p>C) This year, thanks to Daniel Faria, we updated the SEALS client to include the new
functionalities introduced in the interactive matching track. We also updated the
client to use the latest libraries which caused some trouble to some Jena developers.
D) This year, due to technical problems, we were missing the SEALS web portal, but
this dis not seem to affect the participation since the number of submitted systems
increased with respect to 2014. In any case, we hope to bring back the SEALS
portal for future OAEI campaigns.</p>
      <p>E) As already proposed in previous years, it would be good to set the preliminary
evaluation results by the end of July to avoid last minute errors and incompatibilities
with the SEALS client.</p>
      <p>F) Again, given the high number of publications on data interlinking, it is surprising
to have so few participants to the instance matching track, although this number
has increased. Nevertheless, we are in direct contact with data interlinking system
developers that may be interested in integrating their benchmarks within the OAEI.
G) As in previous years we had a panel discussion session during the OM workshop
where we discussed about hot topics and future lines for the OAEI. Among others,
we discussed about the need of continuing the effort of improving the interactive
track and adding uncertainty to the OAEI benchmarks (as in the Conference track).
Furthermore we also analyzed the feasibility of joining efforts with the Process
Model Matching Contest (PMMC): https://ai.wu.ac.at/emisa2015/
contest.php. As a first step we planned to make available an interface to
convert from/to a model specification to OWL in order to ease the participation of
OAEI systems in the PMMC and vice versa.</p>
      <p>Here are lessons learned per OAEI 2015 track:
A) Most of the systems participating in the Multifarm track pre-compiles a local
dictionary in order to avoid multiple accesses to the translators within the matching
process which would exceed the allowed (free) translation quota. For future years
we may consider limiting the amount of local information a system can store.
B) In order to attract more instance matching systems to participate in value
semantics (val-sem), value structure (val-struct), and value structure semantics
(val-structsem) tasks, we need to produce benchmarks that have fewer instances (in the order
of 10000), of the same type (in our benchmark we asked systems to compare
instances of different types). To balance those aspects, we must then produce
benchmarks that are more complex i.e., contain more complex transformations.
C) In the largebio track we flagged incoherence-causing mappings (i.e., those removed
by at least one of the used repair approaches: Alcomo [26], LogMap [20] or AML
[31]) by setting their relation to ”?” (unknown). These ”?” mappings are neither
considered as positive nor as negative when evaluating the participating ontology
matching systems, but will simply be ignored. The interactive track uses the
reference alignments of each track to simulate the user interaction or Oracle. This year,
when simulating the user interaction with the largebio dataset, the Oracle returned
“true” when asked about a mapping flagged as “unknown”. However, we realized
that returning true leads to erratic behavior (and loss of performance) for algorithms
computing an interactive repair. Thus, as the role of user feedback during repair is
extremely important, we should ensure that the Oracle’s behavior simulates it in a
sensible manner.</p>
      <p>D) Based on the uncertain reference alignment from the conference track we conclude
that many more matchers provide alignments with a range of confidence values
than in the past which better corresponds to human evaluation of the match quality.
E) In the interactive track we simulate users with different error rates, i.e., given a
query about a mapping there is a random chance that the user is wrong. A “smart”
interactive system could potentially ask the same question several times in order to
mitigate the effect of the simulated error rate of the user. In the future we plan to
extend the SEALS client to identify this potential behavior in interactive matching
systems.</p>
      <p>F) For the OA4QA track, both averaging F-measures and computing it from the
averaged precision and recall values raised confusion while reporting the results. For
the next edition we plan to use a global precision and recall (and consequently
Fmeasure) on the combined result sets of all the query, similarly to what is already
done in the conference track. One major challenge in the design of the new scoring
function is to keep the scoring balanced despite differences in cardinality of the
result sets of the single queries.
12
OAEI 2015 saw an increased number of participants. We hope to keep this trend next
year. Most of the test cases are performed on the SEALS platform, including the
instance matching track. This is good news for the interoperability of matching systems.
The fact that the SEALS platform can be used for such a variety of tasks is also a good
sign of its relevance.</p>
      <p>Again, we observed improvements of runtimes. For example, all systems but two
participating in the anatomy track finished in less than 15 minutes. As usual, most of the
systems favor precision over recall. In general, participating matching systems do not
take advantage of alignment repairing system and return sometimes incoherent
alignments. This is a problem if their result has to be taken as input by a reasoning system.</p>
      <p>This year we also evaluated ontology matching systems in query answering tasks.
The track was not fully based on SEALS but it reused the computed alignments from
the conference track, which runs in the SEALS client. This new track shed light on
the performance of ontology matching systems with respect to the coherence of their
computed alignments.</p>
      <p>A novelty of this year was an extended evaluation in the conference, interactive and
instance matching tracks. This brought interesting insights on the performances of such
systems and should certainly be continued.</p>
      <p>Most of the participants have provided a description of their systems and their
experience in the evaluation. These OAEI papers, like the present one, have not been peer
reviewed. However, they are full contributions to this evaluation exercise and reflect the
hard work and clever insight people put in the development of participating systems.
Reading the papers of the participants should help people involved in ontology
matching to find what makes these algorithms work and what could be improved. Sometimes,
participants offer alternate evaluation results.</p>
      <p>The Ontology Alignment Evaluation Initiative will continue these tests by
improving both test cases and testing methodology for being more accurate. Matching
evaluation still remains a challenging topic, which is worth further research in order to
facilitate the progress of the field [33]. More information can be found at:</p>
    </sec>
    <sec id="sec-10">
      <title>Acknowledgements</title>
      <p>We warmly thank the participants of this campaign. We know that they have worked
hard for having their matching tools executable in time and they provided useful reports
on their experience. The best way to learn about the results remains to read the following
papers.</p>
      <p>We are very grateful to the Universidad Polite´cnica de Madrid (UPM), especially to
Nandana Mihindukulasooriya and Asuncio´n Go´mez Pe´rez, for moving, setting up and
providing the necessary infrastructure to run the SEALS repositories.</p>
      <p>We are also grateful to Martin Ringwald and Terry Hayamizu for providing the
reference alignment for the anatomy ontologies and thank Elena Beisswanger for her
thorough support on improving the quality of the data set.</p>
      <p>We thank Christian Meilicke for his support of the anatomy test case.</p>
      <p>We thank Khiat Abderrahmane for his support in the Arabic data set and Catherine
Comparot for her feedback and support in the MultiFarm test case.</p>
      <p>We also thank for their support the other members of the Ontology Alignment
Evaluation Initiative steering committee: Yannis Kalfoglou (Ricoh laboratories, UK),
Miklos Nagy (The Open University (UK), Natasha Noy (Stanford University, USA),
Yuzhong Qu (Southeast University, CN), York Sure (Leibniz Gemeinschaft, DE), Jie
Tang (Tsinghua University, CN), Heiner Stuckenschmidt (Mannheim Universita¨t, DE),
George Vouros (University of the Aegean, GR).</p>
      <p>Je´roˆme Euzenat, Ernesto Jimenez-Ruiz, and Ca´ssia Trojahn dos Santos have been
partially supported by the SEALS (IST-2009-238975) European project in the previous
years.</p>
      <p>Ernesto has also been partially supported by the Seventh Framework Program (FP7)
of the European Commission under Grant Agreement 318338, “Optique”, the Royal
Society, and the EPSRC projects Score!, DBOnto and MaSI3.</p>
      <p>Ondrˇej Zamazal has been supported by the CSF grant no. 14-14076P.</p>
      <p>Daniel Faria was supported by the Portuguese FCT through the SOMER
project (PTDC/EIA-EIA/119119/2010), and the LASIGE Strategic Project
(PEstOE/EEI/UI0408/ 2015).</p>
      <p>Michelle Cheatham has been supported by the National Science Foundation award
ICER-1440202 “EarthCube Building Blocks: Collaborative Proposal: GeoLink”.
Dmitriy Zheleznyakov, and Ian Horrocks. Ontology based access to exploration data at
statoil. In The Semantic Web - ISWC 2015 - 14th International Semantic Web Conference,
Bethlehem, PA, USA, October 11-15, 2015, Proceedings, Part II, pages 93–112, 2015.
25. Ilianna Kollia, Birte Glimm, and Ian Horrocks. SPARQL query answering over OWL
ontologies. In The Semantic Web: Research and Applications, pages 382–396. Springer, 2011.
26. Christian Meilicke. Alignment Incoherence in Ontology Matching. PhD thesis, University</p>
      <p>Mannheim, 2011.
27. Christian Meilicke, Rau´l Garc´ıa Castro, Frederico Freitas, Willem Robert van Hage, Elena
Montiel-Ponsoda, Ryan Ribeiro de Azevedo, Heiner Stuckenschmidt, Ondrej Sva´b-Zamazal,
Vojtech Sva´tek, Andrei Tamilin, Ca´ssia Trojahn, and Shenghui Wang. MultiFarm: A
benchmark for multilingual ontology matching. Journal of web semantics, 15(3):62–68, 2012.
28. Boris Motik, Rob Shearer, and Ian Horrocks. Hypertableau reasoning for description logics.</p>
      <p>Journal of Artificial Intelligence Research, 36:165–228, 2009.
29. Heiko Paulheim, Sven Hertling, and Dominique Ritze. Towards evaluating interactive
ontology matching tools. In Proc. 10th Extended Semantic Web Conference (ESWC), Montpellier
(FR), pages 31–45, 2013.
30. Catia Pesquita, Daniel Faria, Emanuel Santos, and Francisco Couto. To repair or not to
repair: reconciling correctness and coherence in ontology reference alignments. In Proc. 8th
ISWC ontology matching workshop (OM), Sydney (AU), page this volume, 2013.
31. Emanuel Santos, Daniel Faria, Catia Pesquita, and Francisco Couto. Ontology alignment
repair through modularization and confidence-based heuristics. CoRR, abs/1307.5322, 2013.
32. Tzanina Saveta, Evangelia Daskalaki, Giorgos Flouris, Irini Fundulaki, Melanie Herschel,
and Axel-Cyrille Ngonga Ngomo. Pushing the limits of instance matching systems: A
semantics-aware benchmark for linked data. In WWW, Companion Volume, 2015.
33. Pavel Shvaiko and Je´roˆme Euzenat. Ontology matching: state of the art and future challenges.</p>
      <p>IEEE Transactions on Knowledge and Data Engineering, 25(1):158–176, 2013.
34. Alessandro Solimando, Ernesto Jime´nez-Ruiz, and Giovanna Guerrini. Detecting and
correcting conservativity principle violations in ontology-to-ontology mappings. In The
Semantic Web–ISWC 2014, pages 1–16. Springer, 2014.
35. Alessandro Solimando, Ernesto Jime´nez-Ruiz, and Giovanna Guerrini. Detecting and
Correcting Conservativity Principle Violations in Ontology-to-Ontology Mappings. In
International Semantic Web Conference, 2014.
36. Alessandro Solimando, Ernesto Jime´nez-Ruiz, and Giovanna Guerrini. A multi-strategy
approach for detecting and correcting conservativity principle violations in ontology
alignments. In Proceedings of the 11th International Workshop on OWL: Experiences and
Directions (OWLED 2014) co-located with 13th International Semantic Web Conference on
(ISWC 2014), Riva del Garda, Italy, October 17-18, 2014., pages 13–24, 2014.
37. Alessandro Solimando, Ernesto Jime´nez-Ruiz, and Christoph Pinkel. Evaluating ontology
alignment systems in query answering tasks. In Proceedings of the ISWC 2014 Posters &amp;
Demonstrations Track a track within the 13th International Semantic Web Conference, ISWC
2014, Riva del Garda, Italy, October 21, 2014., pages 301–304, 2014.
38. York Sure, Oscar Corcho, Je´roˆme Euzenat, and Todd Hughes, editors. Proc. ISWC Workshop
on Evaluation of Ontology-based Tools (EON), Hiroshima (JP), 2004.</p>
    </sec>
  </body>
  <back>
    <ref-list>
      <ref id="ref1">
        <mixed-citation>
          1. Jose´ Luis Aguirre, Bernardo Cuenca Grau, Kai Eckert, Je´roˆ me Euzenat, Alfio Ferrara, Robert Willem van Hague,
          <string-name>
            <surname>Laura Hollink</surname>
          </string-name>
          , Ernesto Jime´
          <article-title>nez-</article-title>
          <string-name>
            <surname>Ruiz</surname>
            ,
            <given-names>Christian</given-names>
          </string-name>
          <string-name>
            <surname>Meilicke</surname>
          </string-name>
          , Andriy Nikolov, Dominique Ritze, Franc¸ois Scharffe, Pavel Shvaiko, Ondrej Sva´
          <fpage>b</fpage>
          -Zamazal, Ca´ssia Trojahn, and
          <string-name>
            <given-names>Benjamin</given-names>
            <surname>Zapilko</surname>
          </string-name>
          .
          <article-title>Results of the ontology alignment evaluation initiative 2012</article-title>
          .
          <source>In Proc. 7th ISWC ontology matching workshop (OM)</source>
          , Boston (MA US), pages
          <fpage>73</fpage>
          -
          <lpage>115</lpage>
          ,
          <year>2012</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref2">
        <mixed-citation>
          2.
          <string-name>
            <given-names>Benhamin</given-names>
            <surname>Ashpole</surname>
          </string-name>
          , Marc Ehrig, Je´r oˆme Euzenat, and Heiner Stuckenschmidt, editors.
          <source>Proc. K-Cap Workshop on Integrating Ontologies</source>
          ,
          <source>Banff (Canada)</source>
          ,
          <year>2005</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref3">
        <mixed-citation>
          3.
          <string-name>
            <given-names>Olivier</given-names>
            <surname>Bodenreider</surname>
          </string-name>
          .
          <article-title>The unified medical language system (UMLS): integrating biomedical terminology</article-title>
          .
          <source>Nucleic Acids Research</source>
          ,
          <volume>32</volume>
          :
          <fpage>267</fpage>
          -
          <lpage>270</lpage>
          ,
          <year>2004</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref4">
        <mixed-citation>
          4.
          <string-name>
            <given-names>Caterina</given-names>
            <surname>Caracciolo</surname>
          </string-name>
          , Je´roˆ me Euzenat, Laura Hollink, Ryutaro Ichise, Antoine Isaac, Ve´ronique Malaise´,
          <string-name>
            <surname>Christian</surname>
            <given-names>Meilicke</given-names>
          </string-name>
          , Juan Pane, Pavel Shvaiko, Heiner Stuckenschmidt, Ondrej Sva´
          <article-title>b-</article-title>
          <string-name>
            <surname>Zamazal</surname>
          </string-name>
          ,
          <article-title>and Vojtech Sva´tek. Results of the ontology alignment evaluation initiative 2008</article-title>
          .
          <source>In Proc. 3rd ISWC ontology matching workshop (OM)</source>
          ,
          <source>Karlsruhe (DE)</source>
          , pages
          <fpage>73</fpage>
          -
          <lpage>120</lpage>
          ,
          <year>2008</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref5">
        <mixed-citation>
          5.
          <string-name>
            <given-names>Michelle</given-names>
            <surname>Cheatham</surname>
          </string-name>
          and
          <string-name>
            <given-names>Pascal</given-names>
            <surname>Hitzler</surname>
          </string-name>
          .
          <source>Conference v2</source>
          .
          <article-title>0: An uncertain version of the oaei conference benchmark</article-title>
          .
          <source>In The Semantic Web-ISWC</source>
          <year>2014</year>
          , pages
          <fpage>33</fpage>
          -
          <lpage>48</lpage>
          . Springer,
          <year>2014</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref6">
        <mixed-citation>
          6.
          <string-name>
            <given-names>Bernardo</given-names>
            <surname>Cuenca</surname>
          </string-name>
          <string-name>
            <surname>Grau</surname>
          </string-name>
          , Zlatan Dragisic, Kai Eckert, Je´roˆ me Euzenat, Alfio Ferrara, Roger Granada, Valentina Ivanova, Ernesto Jime´
          <fpage>nez</fpage>
          -Ruiz, Andreas Oskar Kempf, Patrick Lambrix, Andriy Nikolov, Heiko Paulheim, Dominique Ritze, Franc¸ois Scharffe, Pavel Shvaiko, Ca´ssia Trojahn dos Santos, and
          <string-name>
            <given-names>Ondrej</given-names>
            <surname>Zamazal</surname>
          </string-name>
          .
          <article-title>Results of the ontology alignment evaluation initiative 2013</article-title>
          . In Pavel Shvaiko, Je´r oˆme Euzenat, Kavitha Srinivas, Ming Mao, and Ernesto Jime´
          <fpage>nez</fpage>
          -Ruiz, editors,
          <source>Proc. 8th ISWC workshop on ontology matching (OM)</source>
          ,
          <source>Sydney (NSW AU)</source>
          , pages
          <fpage>61</fpage>
          -
          <lpage>100</lpage>
          ,
          <year>2013</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref7">
        <mixed-citation>
          7.
          <string-name>
            <given-names>Jim</given-names>
            <surname>Dabrowski</surname>
          </string-name>
          and
          <string-name>
            <given-names>Ethan V.</given-names>
            <surname>Munson</surname>
          </string-name>
          .
          <article-title>40 years of searching for the best computer system response time</article-title>
          .
          <source>Interacting with Computers</source>
          ,
          <volume>23</volume>
          (
          <issue>5</issue>
          ):
          <fpage>555</fpage>
          -
          <lpage>564</lpage>
          ,
          <year>2011</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref8">
        <mixed-citation>
          8. Je´r oˆme David, Je´roˆ me Euzenat, Franc¸ois Scharffe, and Ca´ssia Trojahn dos Santos.
          <source>The alignment API 4.0. Semantic web journal</source>
          ,
          <volume>2</volume>
          (
          <issue>1</issue>
          ):
          <fpage>3</fpage>
          -
          <lpage>10</lpage>
          ,
          <year>2011</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref9">
        <mixed-citation>
          9.
          <string-name>
            <given-names>Zlatan</given-names>
            <surname>Dragisic</surname>
          </string-name>
          , Kai Eckert, Je´roˆ me Euzenat, Daniel Faria, Alfio Ferrara, Roger Granada, Valentina Ivanova, Ernesto Jime´
          <fpage>nez</fpage>
          -Ruiz, Andreas Oskar Kempf, Patrick Lambrix, Stefano Montanelli, Heiko Paulheim, Dominique Ritze, Pavel Shvaiko, Alessandro Solimando,
          <article-title>Ca´ssia Trojahn dos Santos, Ondrej Zamazal, and Bernardo Cuenca Grau. Results of the ontology alignment evaluation initiative 2014</article-title>
          .
          <source>In Proceedings of the 9th International Workshop on Ontology Matching collocated with the 13th International Semantic Web Conference (ISWC</source>
          <year>2014</year>
          ),
          <source>Riva del Garda</source>
          , Trentino, Italy, October
          <volume>20</volume>
          ,
          <year>2014</year>
          ., pages
          <fpage>61</fpage>
          -
          <lpage>104</lpage>
          ,
          <year>2014</year>
          .
        </mixed-citation>
      </ref>
    </ref-list>
  </back>
</article>